diff --git a/.bazelrc b/.bazelrc index bfca26683..962617fcf 100644 --- a/.bazelrc +++ b/.bazelrc @@ -26,6 +26,23 @@ build --per_file_copt=mobile_back_tflite/cpp/backend_tflite/llm_pipeline.cc@-fpe # tools resolved at fetch time still work. build --incompatible_strict_action_env +# Abseil does not survive clang's layering check when it is built for the exec +# configuration, i.e. as part of a host tool such as protoc: +# +# absl/status/statusor.h:58:10: error: module com_google_absl//absl/status: +# statusor does not depend on a module exporting 'absl/types/span.h' +# +# The dependency is declared -- //absl/types:span is in statusor's deps and its +# .cppmap is on the command line -- so this is the runner's clang disagreeing +# with Abseil's own module layering, not a missing edge we could add. It is a +# hygiene diagnostic over third-party code that produces no artifact, so it is +# turned off rather than worked around. +# +# It went unnoticed until the LiteRT 2.2.0 pin move: the action was a remote +# cache hit on every previous green run, and only the cold cache the new pins +# forced actually compiled it. Target configurations keep the check. +build --host_features=-layering_check + # This flag is required by tensorflow common --experimental_repo_remote_exec diff --git a/.github/workflows/ios-build-test-macos.yml b/.github/workflows/ios-build-test-macos.yml index b9b321e49..ca8cc6b56 100644 --- a/.github/workflows/ios-build-test-macos.yml +++ b/.github/workflows/ios-build-test-macos.yml @@ -41,9 +41,15 @@ jobs: - backend: "tflite" with_apple: 0 with_tflite: 1 + with_litert: 0 - backend: "apple" with_apple: 1 with_tflite: 0 + with_litert: 0 + - backend: "litert" + with_apple: 0 + with_tflite: 0 + with_litert: 1 env: PERF_TEST: true BAZEL_OUTPUT_ROOT_ARG: "--output_user_root=/tmp/bazel_output" @@ -55,6 +61,7 @@ jobs: FLUTTER_IOS_TEST_PACKAGE: ios_tests-${{ matrix.backend }}-${{ needs.computed.outputs.build_number }}.zip WITH_APPLE: ${{ matrix.with_apple }} WITH_TFLITE: ${{ matrix.with_tflite }} + WITH_LITERT: ${{ matrix.with_litert }} WITH_PIXEL: 0 WITH_MEDIATEK: 0 WITH_QTI: 0 @@ -229,6 +236,9 @@ jobs: runs-on: ubuntu-22.04 permissions: contents: read + # The litert row runs the llm-* benchmarks, whose go-button check validates + # every delegate choice's model file (two ~1.2 GB LLM exports), so it needs + # the same headroom as the Android unified job. timeout-minutes: 135 # BrowserStack allows only 2 parallel sessions on this account, and a # superseded run's device tests starve the current one -- this job has @@ -254,6 +264,8 @@ jobs: device: "iPhone 16 Pro-18" - backend: "apple" device: "iPhone 16 Pro-18" + - backend: "litert" + device: "iPhone 16 Pro-18" env: TEST_PACKAGE_NAME: ios_tests-${{ matrix.backend }}-${{ needs.computed.outputs.build_number }}.zip BROWSERSTACK_LOGS_DIR: /tmp/browserstack-device-logs @@ -282,11 +294,19 @@ jobs: BROWSERSTACK_DEVICES: >- ["${{ matrix.device }}"] with: - timeout_minutes: 60 + # This is the timeout that actually fires: the job-level one never + # gets a chance, because the step waits out retry_wait_seconds even + # when it will not retry (retry_on_exit_code only covers a trigger + # rejection). Keep it below the job cap minus retry_wait_seconds. + # BrowserStack reports a build as "running" while it is still queued + # for a device, so this budget covers queue time as well as the run. + timeout_minutes: 80 # Exit 9 is a BrowserStack trigger failure, which is what # BROWSERSTACK_ALL_PARALLELS_IN_USE produces when the account's # session budget is busy - retrying it just waits for a free slot. - # Every other failure still fails fast. + # Every other failure still fails fast. A rejection returns + # immediately, but each retry still costs retry_wait_seconds, so a + # long queue can burn into the job cap above. max_attempts: 8 retry_wait_seconds: 300 retry_on_exit_code: 9 diff --git a/.github/workflows/scripts/browserstack-app-automate.sh b/.github/workflows/scripts/browserstack-app-automate.sh index 9ef061037..4e38d8a51 100644 --- a/.github/workflows/scripts/browserstack-app-automate.sh +++ b/.github/workflows/scripts/browserstack-app-automate.sh @@ -125,16 +125,27 @@ download_device_logs() { if [[ -n "$test_id" && "$test_id" != "null" && -n "$device_log_url" && "$device_log_url" != "null" ]]; then echo "Found last test case $test_id with device log URL" - # Download device logs using the extracted URL + # Download device logs using the extracted URL. -f so an HTTP error is a + # failure instead of an error page written to the log, and retries because + # the log is not always served the instant the session ends. local log_file="$LOGS_DIR/${test_id}.log" echo "Downloading device log to $log_file" - curl -s -u "$CREDENTIALS" -X GET "$device_log_url" -o "$log_file" - - if [ -f "$log_file" ]; then - echo "Device logs downloaded successfully to $log_file" + if curl -sf --retry 3 --retry-delay 5 -u "$CREDENTIALS" -X GET \ + "$device_log_url" -o "$log_file" && [ -s "$log_file" ]; then + echo "Device logs downloaded successfully to $log_file ($(wc -c <"$log_file") bytes)" else - echo "Failed to download device logs for test case $test_id" + # Test -s, not -f: a zero-byte log is what this used to leave behind, and + # it reads as "the run produced no output" rather than "the download + # failed". Leave a note in its place so the artifact explains itself -- + # if-no-files-found on the upload step only catches a missing file. + echo "Failed to download device logs for test case $test_id" \ + | tee "$log_file" + echo " session: $session_id" >> "$log_file" + echo " url: $device_log_url" >> "$log_file" fi + else + echo "No device log URL in the session response for build $build_id" \ + | tee "$LOGS_DIR/no-device-log-${session_id}.txt" fi } diff --git a/WORKSPACE b/WORKSPACE index 20414cd79..8a2342a53 100644 --- a/WORKSPACE +++ b/WORKSPACE @@ -30,7 +30,7 @@ http_archive( # rules_python that XLA now brings in: rules_apple's plisttool is generated # from a bootstrap template containing %interpreter_args%, which the older # rules leave unsubstituted, so it fails to parse as Python. These are the -# same versions LiteRT 2.1.5 pins for this dependency set. +# same versions LiteRT 2.2.0 pins for this dependency set. http_archive( name = "build_bazel_rules_apple", sha256 = "a78f26c22ac8d6e3f3fcaad50eace4d9c767688bd7254b75bdf4a6735b299f6a", @@ -51,6 +51,23 @@ http_archive( url = "https://github.com/bazelbuild/apple_support/releases/download/1.23.1/apple_support.1.23.1.tar.gz", ) +# The rules_cc rules_python's py_repositories() would bring in anyway, patched. +# Declared here so it is fetched with the patch rather than without it; the +# declaration below it is a maybe(), so ours wins. +# +# Moving TensorFlow to LiteRT 2.2.0's commit pulled this chain -- XLA, then +# rules_ml_toolchain, then rules_python -- forward to a rules_cc whose +# use_cc_toolchain() is mandatory. That breaks the Android build during +# analysis. See patches/rules_cc_optional_cc_toolchain.patch for the detail. +http_archive( + name = "rules_cc", + patch_args = ["-p1"], + patches = ["//patches:rules_cc_optional_cc_toolchain.patch"], + sha256 = "b8b918a85f9144c01f6cfe0f45e4f2838c7413961a8ff23bc0c6cdf8bb07a3b6", + strip_prefix = "rules_cc-0.1.5", + url = "https://github.com/bazelbuild/rules_cc/releases/download/0.1.5/rules_cc-0.1.5.tar.gz", +) + http_archive( name = "bazel_features", sha256 = "c26b4e69cf02fea24511a108d158188b9d8174426311aac59ce803a78d107648", @@ -131,21 +148,21 @@ http_archive( patch_args = ["-p1"], patches = [ # Add patches for adding png in tflite evaluation code + # Channel detection and the unsigned-char decode buffer, formerly + # png-with-number-of-channels-detected.patch and use_unsigned_char.patch, + # are folded into this one. "//:flutter/third_party/enable-png-in-tensorflow-lite-tools-evaluation.patch", - "//:flutter/third_party/png-with-number-of-channels-detected.patch", - "//:flutter/third_party/use_unsigned_char.patch", # Fix tensorflow not being able to read image files on Windows "//:flutter/third_party/tensorflow-fix-file-opening-mode-for-Windows.patch", + "//:patches/litert_logistic_fp16_msvc.patch", "//:patches/litert-internal-visibility.diff", # Fix for LiteRT crashing on close when using OpenCL accelerator "//:patches/custom_buffer_teardown.patch", - # CoreML delegate calls RepeatedField::resize, which does not exist - "//:patches/litert_coreml_repeatedfield_resize.patch", ], - sha256 = "7d0313c4851deb18af6f5f2dbc002bf01293583b87b819b0949ee33dcfe2d91b", - strip_prefix = "LiteRT-2.1.5", + sha256 = "6d2ce16738199adc5a3cdde76c3c6a6dac636d3b52a1d7790ea524fb0d59f7fc", + strip_prefix = "LiteRT-2.2.0", urls = [ - "https://github.com/google-ai-edge/LiteRT/archive/v2.1.5.tar.gz", + "https://github.com/google-ai-edge/LiteRT/archive/v2.2.0.tar.gz", ], ) @@ -161,13 +178,17 @@ tensorflow_source_repo( patches = [ "//:flutter/third_party/tf-eigen.patch", "//patches:tf_coreml_repeatedfield_resize.patch", + "//patches:tf_logistic_fp16_msvc.patch", "//patches:tf_nnapi_no_mmap_sharing.patch", "//patches:tf_portable_no_onednn_env_vars.patch", ] + PATCH_FILE, - sha256 = "879cf25692d50c60315a4dd3929dccd923d4c44a2c4b95ebb483666d2c16a22a", - strip_prefix = "tensorflow-6d40c20cdfe385746c31da6227b95722f5ece342", + # The commit LiteRT 2.2.0 pins. LiteRT's GPU delegate needs + # @com_google_absl//absl/status:status_macros, which only exists in the + # Abseil this TensorFlow's XLA brings in, so the two move together. + sha256 = "c3c552414ab2e59e72511a21c1df566346a7c8f160909325edec6d1ff403d69d", + strip_prefix = "tensorflow-bcdab1a62e138c8f8784a7477c0be8af6dd0bd0a", urls = [ - "https://github.com/tensorflow/tensorflow/archive/6d40c20cdfe385746c31da6227b95722f5ece342.tar.gz", + "https://github.com/tensorflow/tensorflow/archive/bcdab1a62e138c8f8784a7477c0be8af6dd0bd0a.tar.gz", ], ) diff --git a/flutter/assets/tasks.pbtxt b/flutter/assets/tasks.pbtxt index 7e1367551..9f366cf12 100644 --- a/flutter/assets/tasks.pbtxt +++ b/flutter/assets/tasks.pbtxt @@ -417,7 +417,13 @@ task { quick { min_query_count: 10 min_duration: 10 - max_duration: 40 + # 40s could never satisfy min_query_count on a 1B model: measured on an + # iPhone 16 Pro, llm-1b runs about 10.1s per query on the Metal delegate + # (24.3 tok/s), so ten queries need ~101s and the run ended invalid -- + # isMinQueryMet false -- on both the CPU and the GPU delegate. A run + # stops as soon as min_duration AND min_query_count are both met, so + # this ceiling only costs time on devices that were failing anyway. + max_duration: 150 } rapid { min_query_count: 6 @@ -477,7 +483,13 @@ task { quick { min_query_count: 10 min_duration: 10 - max_duration: 40 + # 40s could never satisfy min_query_count on a 1B model: measured on an + # iPhone 16 Pro, llm-1b runs about 10.1s per query on the Metal delegate + # (24.3 tok/s), so ten queries need ~101s and the run ended invalid -- + # isMinQueryMet false -- on both the CPU and the GPU delegate. A run + # stops as soon as min_duration AND min_query_count are both met, so + # this ceiling only costs time on devices that were failing anyway. + max_duration: 150 } rapid { min_query_count: 6 @@ -657,7 +669,13 @@ task { quick { min_query_count: 10 min_duration: 10 - max_duration: 40 + # 40s could never satisfy min_query_count on a 1B model: measured on an + # iPhone 16 Pro, llm-1b runs about 10.1s per query on the Metal delegate + # (24.3 tok/s), so ten queries need ~101s and the run ended invalid -- + # isMinQueryMet false -- on both the CPU and the GPU delegate. A run + # stops as soon as min_duration AND min_query_count are both met, so + # this ceiling only costs time on devices that were failing anyway. + max_duration: 150 } rapid { min_query_count: 6 @@ -717,7 +735,13 @@ task { quick { min_query_count: 10 min_duration: 10 - max_duration: 40 + # 40s could never satisfy min_query_count on a 1B model: measured on an + # iPhone 16 Pro, llm-1b runs about 10.1s per query on the Metal delegate + # (24.3 tok/s), so ten queries need ~101s and the run ended invalid -- + # isMinQueryMet false -- on both the CPU and the GPU delegate. A run + # stops as soon as min_duration AND min_query_count are both met, so + # this ceiling only costs time on devices that were failing anyway. + max_duration: 150 } rapid { min_query_count: 6 @@ -777,7 +801,13 @@ task { quick { min_query_count: 10 min_duration: 10 - max_duration: 40 + # 40s could never satisfy min_query_count on a 1B model: measured on an + # iPhone 16 Pro, llm-1b runs about 10.1s per query on the Metal delegate + # (24.3 tok/s), so ten queries need ~101s and the run ended invalid -- + # isMinQueryMet false -- on both the CPU and the GPU delegate. A run + # stops as soon as min_duration AND min_query_count are both met, so + # this ceiling only costs time on devices that were failing anyway. + max_duration: 150 } rapid { min_query_count: 6 @@ -837,7 +867,13 @@ task { quick { min_query_count: 10 min_duration: 10 - max_duration: 40 + # 40s could never satisfy min_query_count on a 1B model: measured on an + # iPhone 16 Pro, llm-1b runs about 10.1s per query on the Metal delegate + # (24.3 tok/s), so ten queries need ~101s and the run ended invalid -- + # isMinQueryMet false -- on both the CPU and the GPU delegate. A run + # stops as soon as min_duration AND min_query_count are both met, so + # this ceiling only costs time on devices that were failing anyway. + max_duration: 150 } rapid { min_query_count: 6 diff --git a/flutter/cpp/datasets/ifeval_utils/BUILD b/flutter/cpp/datasets/ifeval_utils/BUILD index 70f3962ce..731e6163f 100644 --- a/flutter/cpp/datasets/ifeval_utils/BUILD +++ b/flutter/cpp/datasets/ifeval_utils/BUILD @@ -40,3 +40,13 @@ cc_library( "@oleander_stemming_library", ], ) + +cc_test( + name = "common_test", + srcs = ["common_test.cc"], + linkstatic = 1, + deps = [ + ":ifeval_utils", + "@com_google_googletest//:gtest", + ], +) diff --git a/flutter/cpp/datasets/ifeval_utils/common.h b/flutter/cpp/datasets/ifeval_utils/common.h index bba83ed4b..970976dbe 100644 --- a/flutter/cpp/datasets/ifeval_utils/common.h +++ b/flutter/cpp/datasets/ifeval_utils/common.h @@ -54,7 +54,10 @@ inline bool contains_string(const std::string& text, inline bool ends_with(const std::string& s, const std::string& suf, unsigned threshold) { if (s.size() < suf.size()) return false; - std::string a = tolower(s.substr(s.size() - (suf.size() + threshold))); + // The window may be longer than the response itself, in which case it is the + // whole response; subtracting unclamped would wrap and make substr throw. + const std::size_t window = std::min(s.size(), suf.size() + threshold); + std::string a = tolower(s.substr(s.size() - window)); std::string b = tolower(suf); return threshold == 0 ? a == b : contains_string(a, b); } @@ -142,8 +145,10 @@ inline std::string remove_font_modifiers(const std::string& s) { } // skip emphasis/strong/strike/escape chars as long as they're not preceeded - // by an escape character - if ((c == '*' || c == '_' || c == '~' || c == '\\') && s[i - 1] != '\\') + // by an escape character. The first character has nothing before it, so it + // is never escaped -- without the i == 0 guard this reads s[SIZE_MAX]. + if ((c == '*' || c == '_' || c == '~' || c == '\\') && + (i == 0 || s[i - 1] != '\\')) continue; // remove heading markers (#) at line starts diff --git a/flutter/cpp/datasets/ifeval_utils/common_test.cc b/flutter/cpp/datasets/ifeval_utils/common_test.cc new file mode 100644 index 000000000..58f40e44a --- /dev/null +++ b/flutter/cpp/datasets/ifeval_utils/common_test.cc @@ -0,0 +1,79 @@ +/* Copyright 2026 The MLPerf Authors. All Rights Reserved. + +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + +Unless required by applicable law or agreed to in writing, software +distributed under the License is distributed on an "AS IS" BASIS, +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +See the License for the specific language governing permissions and +limitations under the License. +==============================================================================*/ + +#include "flutter/cpp/datasets/ifeval_utils/common.h" + +#include + +#include "gtest/gtest.h" + +namespace mlperf { +namespace mobile { +namespace ifeval { +namespace { + +// A response that opens with a markdown modifier used to read s[i - 1] with +// i == 0, i.e. index SIZE_MAX. Under -O2 that crashed the accuracy pass with +// EXC_BAD_ACCESS at 0xffffffffffffffff. +TEST(RemoveFontModifiers, StripsALeadingModifier) { + EXPECT_EQ(remove_font_modifiers("*Hello* world"), "Hello world"); + EXPECT_EQ(remove_font_modifiers("_x_"), "x"); + EXPECT_EQ(remove_font_modifiers("~y~"), "y"); + EXPECT_EQ(remove_font_modifiers("\\z"), "z"); +} + +TEST(RemoveFontModifiers, KeepsAnEscapedModifier) { + EXPECT_EQ(remove_font_modifiers("a\\*b"), "a*b"); +} + +TEST(RemoveFontModifiers, HandlesTheRemainingMarkup) { + EXPECT_EQ(remove_font_modifiers(""), ""); + EXPECT_EQ(remove_font_modifiers("`code`"), "code"); + EXPECT_EQ(remove_font_modifiers("# Title"), " Title"); + EXPECT_EQ(remove_font_modifiers("> quote"), " quote"); +} + +TEST(TransformResponse, AppliesTheFontMask) { + EXPECT_EQ(transform_response("*starts with emphasis*")[1], + "starts with emphasis"); +} + +// ends_with looks at the last suf.size() + threshold characters. When the +// response is shorter than that window the subtraction used to wrap, and +// substr threw std::out_of_range. +TEST(EndsWith, AcceptsAResponseShorterThanTheWindow) { + EXPECT_TRUE(ends_with("the end.", "the end.", 3)); + EXPECT_TRUE(ends_with("...and the end.", "the end.", 3)); +} + +TEST(EndsWith, StillRejectsNonMatches) { + EXPECT_FALSE(ends_with("something else", "the end.", 3)); + EXPECT_FALSE(ends_with("short", "a much longer suffix", 3)); +} + +TEST(EndsWith, MatchesExactlyAtThresholdZero) { + EXPECT_TRUE(ends_with("abc", "abc", 0)); + EXPECT_FALSE(ends_with("abcd", "abc", 0)); +} + +} // namespace +} // namespace ifeval +} // namespace mobile +} // namespace mlperf + +int main(int argc, char **argv) { + ::testing::InitGoogleTest(&argc, argv); + return RUN_ALL_TESTS(); +} diff --git a/flutter/cpp/flutter/BUILD b/flutter/cpp/flutter/BUILD index 18cc7b2a6..619c7ac0b 100644 --- a/flutter/cpp/flutter/BUILD +++ b/flutter/cpp/flutter/BUILD @@ -56,8 +56,8 @@ apple_xcframework( "device": ["arm64"], }, minimum_os_versions = { - "ios": "13.1", - "macos": "13.1", + "ios": "15.0", + "macos": "15.0", }, deps = [ "//flutter/cpp/flutter:bridge", diff --git a/flutter/integration_test/expected_accuracy.dart b/flutter/integration_test/expected_accuracy.dart index a502e7776..27fecf834 100644 --- a/flutter/integration_test/expected_accuracy.dart +++ b/flutter/integration_test/expected_accuracy.dart @@ -23,7 +23,11 @@ const Map _imageClassificationV2 = { // LiteRT vision runs fp32 models on the GPU accelerator (CPU fallback) // on the Pixel 10 Pro CI job. Wide intervals spanning the other // accelerators' values until the first runs produce numbers. + // The iOS LiteRT job selects the CPU delegate, which runs the *quantized* + // exports; the bare 'cpu' interval was measured on the Windows TFLite fp32 + // model, so LiteRT needs its own key here too. 'gpu|LiteRT': Interval(min: 0.75, max: 0.90), + 'cpu|LiteRT': Interval(min: 0.75, max: 0.90), 'tpu': Interval(min: 0.82, max: 0.84), 'ane': Interval(min: 0.69, max: 0.91), 'cpu&gpu&ane': Interval(min: 0.69, max: 0.91), @@ -37,6 +41,7 @@ const Map _imageClassificationOfflineV2 = { 'cpu': Interval(min: 0.88, max: 0.91), 'npu': Interval(min: 0.69, max: 0.71), 'gpu|LiteRT': Interval(min: 0.75, max: 0.95), + 'cpu|LiteRT': Interval(min: 0.75, max: 0.95), 'tpu': Interval(min: 0.89, max: 0.91), 'ane': Interval(min: 0.69, max: 0.91), 'cpu&gpu&ane': Interval(min: 0.69, max: 0.91), @@ -50,6 +55,7 @@ const Map _objectDetection = { 'cpu': Interval(min: 0.31, max: 0.32), 'npu': Interval(min: 0.28, max: 0.31), 'gpu|LiteRT': Interval(min: 0.25, max: 0.45), + 'cpu|LiteRT': Interval(min: 0.25, max: 0.45), 'tpu': Interval(min: 0.36, max: 0.38), 'ane|TFLite': Interval(min: 0.31, max: 0.34), 'ane|Core ML': Interval(min: 0.45, max: 0.46), @@ -64,6 +70,7 @@ const Map _imageSegmentationV2 = { 'cpu': Interval(min: 0.38, max: 0.40), 'npu': Interval(min: 0.33, max: 0.34), 'gpu|LiteRT': Interval(min: 0.30, max: 0.45), + 'cpu|LiteRT': Interval(min: 0.30, max: 0.45), 'tpu': Interval(min: 0.33, max: 0.34), 'ane|TFLite': Interval(min: 0.38, max: 0.40), 'ane|Core ML': Interval(min: 0.38, max: 0.40), @@ -79,6 +86,7 @@ const Map _naturalLanguageProcessing = { 'tpu': Interval(min: 1.00, max: 1.00), 'gpu|TFLite': Interval(min: 1.00, max: 1.00), 'gpu|LiteRT': Interval(min: 0.80, max: 1.00), + 'cpu|LiteRT': Interval(min: 0.80, max: 1.00), // 1.00 in simulator, 0.80 on iphone 12 mini 'gpu|Core ML': Interval(min: 0.80, max: 1.00), 'cpu&gpu&ane': Interval(min: 0.80, max: 1.00), @@ -92,6 +100,7 @@ const Map _superResolution = { 'cpu': Interval(min: 0.32, max: 0.35), 'npu': Interval(min: 0.32, max: 0.35), 'gpu|LiteRT': Interval(min: 0.30, max: 0.40), + 'cpu|LiteRT': Interval(min: 0.30, max: 0.40), 'tpu': Interval(min: 0.32, max: 0.35), 'ane|TFLite': Interval(min: 0.32, max: 0.35), 'ane|Core ML': Interval(min: 0.32, max: 0.35), diff --git a/flutter/ios/Runner.xcodeproj/project.pbxproj b/flutter/ios/Runner.xcodeproj/project.pbxproj index 1aab76319..7855aa273 100644 --- a/flutter/ios/Runner.xcodeproj/project.pbxproj +++ b/flutter/ios/Runner.xcodeproj/project.pbxproj @@ -9,13 +9,16 @@ /* Begin PBXBuildFile section */ 00168AD425E36DDA00BC9D7D /* CoreML.framework in Frameworks */ = {isa = PBXBuildFile; fileRef = 00168AD325E36DDA00BC9D7D /* CoreML.framework */; }; 00168AD625E36DF700BC9D7D /* libc++.tbd in Frameworks */ = {isa = PBXBuildFile; fileRef = 00168AD525E36DEC00BC9D7D /* libc++.tbd */; }; + 022193EFD2A6FAE8C1871FE5 /* liblitertbackend.xcframework in Embed Frameworks */ = {isa = PBXBuildFile; fileRef = A63163BAEACD40DF0640E1BC /* liblitertbackend.xcframework */; settings = {ATTRIBUTES = (CodeSignOnCopy, RemoveHeadersOnCopy, ); }; }; 1498D2341E8E89220040F4C2 /* GeneratedPluginRegistrant.m in Sources */ = {isa = PBXBuildFile; fileRef = 1498D2331E8E89220040F4C2 /* GeneratedPluginRegistrant.m */; }; 19C134532876922700263A48 /* backend_bridge_fw.xcframework in Frameworks */ = {isa = PBXBuildFile; fileRef = 19C1344F2876919E00263A48 /* backend_bridge_fw.xcframework */; }; 19C134542876922700263A48 /* backend_bridge_fw.xcframework in Embed Frameworks */ = {isa = PBXBuildFile; fileRef = 19C1344F2876919E00263A48 /* backend_bridge_fw.xcframework */; settings = {ATTRIBUTES = (CodeSignOnCopy, RemoveHeadersOnCopy, ); }; }; 19C1345E2878151A00263A48 /* libtflitebackend.xcframework in Frameworks */ = {isa = PBXBuildFile; fileRef = 19C1345D2878151A00263A48 /* libtflitebackend.xcframework */; }; 19C1345F2878151A00263A48 /* libtflitebackend.xcframework in Embed Frameworks */ = {isa = PBXBuildFile; fileRef = 19C1345D2878151A00263A48 /* libtflitebackend.xcframework */; settings = {ATTRIBUTES = (CodeSignOnCopy, RemoveHeadersOnCopy, ); }; }; + 28A44D9888F85147505E7FAF /* LiteRtMetalAccelerator.framework in Embed Frameworks */ = {isa = PBXBuildFile; fileRef = B30623E31D389AC786037A26 /* LiteRtMetalAccelerator.framework */; settings = {ATTRIBUTES = (CodeSignOnCopy, RemoveHeadersOnCopy, ); }; }; 2C25ABBE270595F000518142 /* RunnerTests.m in Sources */ = {isa = PBXBuildFile; fileRef = 2C25ABBD270595F000518142 /* RunnerTests.m */; }; 3B3967161E833CAA004F5970 /* AppFrameworkInfo.plist in Resources */ = {isa = PBXBuildFile; fileRef = 3B3967151E833CAA004F5970 /* AppFrameworkInfo.plist */; }; + 62133B0FE9193D87CE5EADC2 /* liblitertbackend.xcframework in Frameworks */ = {isa = PBXBuildFile; fileRef = A63163BAEACD40DF0640E1BC /* liblitertbackend.xcframework */; }; 74858FAF1ED2DC5600515810 /* AppDelegate.swift in Sources */ = {isa = PBXBuildFile; fileRef = 74858FAE1ED2DC5600515810 /* AppDelegate.swift */; }; 97C146FC1CF9000F007C117D /* Main.storyboard in Resources */ = {isa = PBXBuildFile; fileRef = 97C146FA1CF9000F007C117D /* Main.storyboard */; }; 97C146FE1CF9000F007C117D /* Assets.xcassets in Resources */ = {isa = PBXBuildFile; fileRef = 97C146FD1CF9000F007C117D /* Assets.xcassets */; }; @@ -46,6 +49,8 @@ files = ( BD6EDD202884143500106593 /* libcoremlbackend.xcframework in Embed Frameworks */, 19C1345F2878151A00263A48 /* libtflitebackend.xcframework in Embed Frameworks */, + 022193EFD2A6FAE8C1871FE5 /* liblitertbackend.xcframework in Embed Frameworks */, + 28A44D9888F85147505E7FAF /* LiteRtMetalAccelerator.framework in Embed Frameworks */, 19C134542876922700263A48 /* backend_bridge_fw.xcframework in Embed Frameworks */, ); name = "Embed Frameworks"; @@ -84,7 +89,9 @@ 97C146FB1CF9000F007C117D /* Base */ = {isa = PBXFileReference; lastKnownFileType = file.storyboard; name = Base; path = Base.lproj/Main.storyboard; sourceTree = ""; }; 97C146FD1CF9000F007C117D /* Assets.xcassets */ = {isa = PBXFileReference; lastKnownFileType = folder.assetcatalog; path = Assets.xcassets; sourceTree = ""; }; 97C147001CF9000F007C117D /* Base */ = {isa = PBXFileReference; lastKnownFileType = file.storyboard; name = Base; path = Base.lproj/LaunchScreen.storyboard; sourceTree = ""; }; + A63163BAEACD40DF0640E1BC /* liblitertbackend.xcframework */ = {isa = PBXFileReference; lastKnownFileType = wrapper.xcframework; name = liblitertbackend.xcframework; path = frameworks/liblitertbackend.xcframework; sourceTree = ""; }; B2F3E7D2361B9115B41D6B52 /* Pods-Runner.release.xcconfig */ = {isa = PBXFileReference; includeInIndex = 1; lastKnownFileType = text.xcconfig; name = "Pods-Runner.release.xcconfig"; path = "Target Support Files/Pods-Runner/Pods-Runner.release.xcconfig"; sourceTree = ""; }; + B30623E31D389AC786037A26 /* LiteRtMetalAccelerator.framework */ = {isa = PBXFileReference; lastKnownFileType = wrapper.framework; name = LiteRtMetalAccelerator.framework; path = frameworks/LiteRtMetalAccelerator.framework; sourceTree = ""; }; BD6EDD1E2884143500106593 /* libcoremlbackend.xcframework */ = {isa = PBXFileReference; lastKnownFileType = wrapper.xcframework; name = libcoremlbackend.xcframework; path = frameworks/libcoremlbackend.xcframework; sourceTree = ""; }; C225CCAFF2C9DB0836BBE358 /* Pods-Runner.debug.xcconfig */ = {isa = PBXFileReference; includeInIndex = 1; lastKnownFileType = text.xcconfig; name = "Pods-Runner.debug.xcconfig"; path = "Target Support Files/Pods-Runner/Pods-Runner.debug.xcconfig"; sourceTree = ""; }; E8EF4D7B34CC3D1FBA66853A /* Pods_RunnerTests.framework */ = {isa = PBXFileReference; explicitFileType = wrapper.framework; includeInIndex = 0; path = Pods_RunnerTests.framework; sourceTree = BUILT_PRODUCTS_DIR; }; @@ -109,6 +116,7 @@ 00168AD425E36DDA00BC9D7D /* CoreML.framework in Frameworks */, DD2FD2B718B01E010F0F88BB /* Pods_Runner.framework in Frameworks */, 19C1345E2878151A00263A48 /* libtflitebackend.xcframework in Frameworks */, + 62133B0FE9193D87CE5EADC2 /* liblitertbackend.xcframework in Frameworks */, ); runOnlyForDeploymentPostprocessing = 0; }; @@ -120,6 +128,8 @@ children = ( BD6EDD1E2884143500106593 /* libcoremlbackend.xcframework */, 19C1345D2878151A00263A48 /* libtflitebackend.xcframework */, + A63163BAEACD40DF0640E1BC /* liblitertbackend.xcframework */, + B30623E31D389AC786037A26 /* LiteRtMetalAccelerator.framework */, 19C134592876A70900263A48 /* ios.mk */, 19C1344F2876919E00263A48 /* backend_bridge_fw.xcframework */, 19C1344E2876918E00263A48 /* frameworks */, diff --git a/flutter/ios/Runner/AppDelegate.swift b/flutter/ios/Runner/AppDelegate.swift index 626664468..7b9fc9243 100644 --- a/flutter/ios/Runner/AppDelegate.swift +++ b/flutter/ios/Runner/AppDelegate.swift @@ -2,12 +2,18 @@ import Flutter import UIKit @main -@objc class AppDelegate: FlutterAppDelegate { +@objc class AppDelegate: FlutterAppDelegate, FlutterImplicitEngineDelegate { override func application( _ application: UIApplication, didFinishLaunchingWithOptions launchOptions: [UIApplication.LaunchOptionsKey: Any]? ) -> Bool { - GeneratedPluginRegistrant.register(with: self) return super.application(application, didFinishLaunchingWithOptions: launchOptions) } + + // Under the UIScene lifecycle the implicit engine is created after + // didFinishLaunchingWithOptions, so registering plugins there would run + // against an engine that does not exist yet. + func didInitializeImplicitFlutterEngine(_ engineBridge: FlutterImplicitEngineBridge) { + GeneratedPluginRegistrant.register(with: engineBridge.pluginRegistry) + } } diff --git a/flutter/ios/Runner/Info-Debug.plist b/flutter/ios/Runner/Info-Debug.plist index 2f766b88e..9757f4ed7 100644 --- a/flutter/ios/Runner/Info-Debug.plist +++ b/flutter/ios/Runner/Info-Debug.plist @@ -43,6 +43,27 @@ LaunchScreen UIMainStoryboardFile Main + UIApplicationSceneManifest + + UIApplicationSupportsMultipleScenes + + UISceneConfigurations + + UIWindowSceneSessionRoleApplication + + + UISceneClassName + UIWindowScene + UISceneDelegateClassName + FlutterSceneDelegate + UISceneConfigurationName + flutter + UISceneStoryboardFile + Main + + + + UISupportedInterfaceOrientations UIInterfaceOrientationPortrait diff --git a/flutter/ios/Runner/Info-Release.plist b/flutter/ios/Runner/Info-Release.plist index 302096276..8588fc2ca 100644 --- a/flutter/ios/Runner/Info-Release.plist +++ b/flutter/ios/Runner/Info-Release.plist @@ -39,6 +39,27 @@ LaunchScreen UIMainStoryboardFile Main + UIApplicationSceneManifest + + UIApplicationSupportsMultipleScenes + + UISceneConfigurations + + UIWindowSceneSessionRoleApplication + + + UISceneClassName + UIWindowScene + UISceneDelegateClassName + FlutterSceneDelegate + UISceneConfigurationName + flutter + UISceneStoryboardFile + Main + + + + UISupportedInterfaceOrientations UIInterfaceOrientationPortrait diff --git a/flutter/ios/Runner/Runner.entitlements b/flutter/ios/Runner/Runner.entitlements index 299140a12..6078c6cea 100644 --- a/flutter/ios/Runner/Runner.entitlements +++ b/flutter/ios/Runner/Runner.entitlements @@ -8,5 +8,9 @@ production com.apple.security.app-sandbox + com.apple.developer.kernel.increased-memory-limit + + com.apple.developer.kernel.extended-virtual-addressing + diff --git a/flutter/ios/ci_scripts/ci_post_clone.sh b/flutter/ios/ci_scripts/ci_post_clone.sh index 6b911f5bb..a7b7fb88f 100755 --- a/flutter/ios/ci_scripts/ci_post_clone.sh +++ b/flutter/ios/ci_scripts/ci_post_clone.sh @@ -101,6 +101,7 @@ cd "$MC_REPO_HOME"/flutter && flutter precache --ios echo "$MC_LOG_PREFIX ========== Build app ==========" export WITH_TFLITE="${WITH_TFLITE:-0}" export WITH_APPLE="${WITH_APPLE:-1}" +export WITH_LITERT="${WITH_LITERT:-0}" echo "$MC_LOG_PREFIX Build backend and Flutter packages" # The GCS remote cache occasionally fails with "handshake timed out" diff --git a/flutter/ios/ios.mk b/flutter/ios/ios.mk index 3f1fd8ff1..4ed894996 100644 --- a/flutter/ios/ios.mk +++ b/flutter/ios/ios.mk @@ -20,6 +20,14 @@ backend_bridge_ios_zip=${BAZEL_LINKS_PREFIX}bin/flutter/cpp/flutter/backend_brid flutter_ios_fw_dir=flutter/ios/frameworks +# The Metal accelerator ships as a plain dylib, but an iOS app bundle may only +# embed bundles: a bare Mach-O under Frameworks/ is rejected by App Store +# Connect. Wrap it in a framework whose CFBundleExecutable keeps the original +# filename -- LiteRT dlopens the accelerator by joining its runtime library dir +# with that exact name, so the name has to survive the move. +flutter_ios_metal_fw_name=LiteRtMetalAccelerator +flutter_ios_metal_fw_dir=${flutter_ios_fw_dir}/${flutter_ios_metal_fw_name}.framework + .PHONY: flutter/ios/clean flutter/ios/clean: rm -rf flutter/build/ios @@ -27,17 +35,52 @@ flutter/ios/clean: # BAZEL_OUTPUT_ROOT_ARG is set on our Jenkins CI .PHONY: flutter/ios/libs flutter/ios/libs: + ${backend_litert_ios_lib_deps} # --use_top_level_targets_for_symlinks bazel ${BAZEL_OUTPUT_ROOT_ARG} build ${BAZEL_CACHE_ARG} \ --config=ios \ ${backend_bridge_ios_target} \ ${backend_tflite_ios_target} \ - ${backend_coreml_ios_target} + ${backend_coreml_ios_target} \ + ${backend_litert_ios_target} rm -rf ${flutter_ios_fw_dir} unzip -q -o -d ${flutter_ios_fw_dir} ${backend_bridge_ios_zip} unzip -q -o -d ${flutter_ios_fw_dir} ${backend_tflite_ios_zip} unzip -q -o -d ${flutter_ios_fw_dir} ${backend_coreml_ios_zip} + unzip -q -o -d ${flutter_ios_fw_dir} ${backend_litert_ios_zip} + @# LiteRT dlopens the Metal accelerator from the app bundle's Frameworks + @# directory, so it has to be embedded next to the backend frameworks -- + @# as a framework rather than a loose dylib, see flutter_ios_metal_fw_dir. + rm -rf ${flutter_ios_metal_fw_dir} + mkdir -p ${flutter_ios_metal_fw_dir} + cp -f ${backend_litert_ios_file} ${flutter_ios_metal_fw_dir}/ + @# CFBundleExecutable is the dylib's own filename, which is what keeps + @# kLiteRtEnvOptionTagRuntimeLibraryDir + "libLiteRtMetalAccelerator.dylib" + @# resolving. MinimumOSVersion is read out of the dylib instead of written + @# here: it has to be the minimum its Mach-O actually declares, or App Store + @# Connect rejects the bundle as unable to run where the plist says it can. + @# 2.2.0 raised the accelerator from 14.0 to 15.0, and the 14.0 left behind + @# is what ITMS-90208 rejected in build 269. + minos=$$(vtool -show-build ${backend_litert_ios_file} | awk '/minos/ {print $$2; exit}'); \ + [ -n "$$minos" ] || { echo "no LC_BUILD_VERSION minos in ${backend_litert_ios_file}"; exit 1; }; \ + printf '%s\n' \ + '' \ + '' \ + '' \ + '' \ + 'CFBundleDevelopmentRegionen' \ + 'CFBundleExecutable${backend_litert_ios_bin_filename}' \ + 'CFBundleIdentifierorg.mlcommons.litert.metalaccelerator' \ + 'CFBundleInfoDictionaryVersion6.0' \ + 'CFBundleName${flutter_ios_metal_fw_name}' \ + 'CFBundlePackageTypeFMWK' \ + 'CFBundleShortVersionString${backend_litert_version}' \ + 'CFBundleVersion${backend_litert_version}' \ + 'CFBundleSupportedPlatformsiPhoneOS' \ + "MinimumOSVersion$$minos" \ + '' \ + '' > ${flutter_ios_metal_fw_dir}/Info.plist flutter/ios/release: flutter/check-release-env flutter/ios flutter/prepare flutter/ios/ipa diff --git a/flutter/third_party/enable-png-in-tensorflow-lite-tools-evaluation.patch b/flutter/third_party/enable-png-in-tensorflow-lite-tools-evaluation.patch index cdc27b00a..6afc79abb 100644 --- a/flutter/third_party/enable-png-in-tensorflow-lite-tools-evaluation.patch +++ b/flutter/third_party/enable-png-in-tensorflow-lite-tools-evaluation.patch @@ -1,19 +1,21 @@ -From 2d14d34ef780b60f063e48178b30cc7ab043f196 Mon Sep 17 00:00:00 2001 -From: Koan-Sin Tan -Date: Fri, 31 Mar 2023 17:11:28 +0800 -Subject: [PATCH] enable png in tensorflow/lite/tools/evaluation +Subject: [PATCH] enable png in tflite/tools/evaluation ---- - tensorflow/lite/tools/evaluation/stages/BUILD | 2 ++ - .../stages/image_preprocessing_stage.cc | 26 +++++++++++++++++++ - .../stages/image_preprocessing_stage_test.cc | 2 +- - 3 files changed, 29 insertions(+), 1 deletion(-) +Originally by Koan-Sin Tan; regenerated for LiteRT 2.2.0, which moved the +loaders to TfLiteStatus/absl::string_view and made ImageData::data a +std::vector. The channel-count detection and the unsigned-char decode +buffer, which upstream carried as two follow-up patches, are folded in here -- +they only existed separately because they were written at different times. + +Two deliberate changes from the original: the decode buffer is a std::vector +rather than a bare new[] that was never freed, and a decode failure returns +kTfLiteError instead of aborting the process through CHECK. The test-file hunk +is dropped: it repointed the unit test at grace_hopper.png, which does not +exist in the tree. diff --git a/tflite/tools/evaluation/stages/BUILD b/tflite/tools/evaluation/stages/BUILD -index 9f649588145..e81b284709c 100644 --- a/tflite/tools/evaluation/stages/BUILD +++ b/tflite/tools/evaluation/stages/BUILD -@@ -59,9 +59,11 @@ cc_library( +@@ -61,9 +61,11 @@ ] + select({ "@org_tensorflow//tensorflow:android": [ "@org_tensorflow//tensorflow/core:portable_jpeg_internal", @@ -26,67 +28,73 @@ index 9f649588145..e81b284709c 100644 }), ) diff --git a/tflite/tools/evaluation/stages/image_preprocessing_stage.cc b/tflite/tools/evaluation/stages/image_preprocessing_stage.cc -index a1418c3bcb6..a9750141b3d 100644 --- a/tflite/tools/evaluation/stages/image_preprocessing_stage.cc +++ b/tflite/tools/evaluation/stages/image_preprocessing_stage.cc -@@ -29,4 +29,5 @@ limitations under the License. - #include "absl/strings/ascii.h" +@@ -34,6 +34,7 @@ + #include "absl/strings/string_view.h" #include "jpeglib.h" // from @libjpeg_turbo #include "tensorflow/core/lib/jpeg/jpeg_mem.h" +#include "tensorflow/core/lib/png/png_io.h" - #include "tensorflow/core/platform/logging.h" -@@ -111,6 +112,29 @@ inline void LoadImageJpeg(std::string* filename, ImageData* image_data) { - image_data->data.reset(float_image); + #include "tflite/c/c_api_types.h" + #include "tflite/c/common.h" + #include "tflite/kernels/internal/reference/pad.h" +@@ -142,6 +143,48 @@ + return kTfLiteOk; } +// Loads the png image. -+inline void LoadImagePng(std::string* filename, ImageData* image_data) { ++inline TfLiteStatus LoadImagePng(absl::string_view filename, ++ ImageData* image_data) { + // Reads image. -+ std::ifstream t(*filename, std::ios::binary); ++ std::ifstream t(std::string(filename).c_str(), ++ std::ios::in | std::ios::binary); ++ if (!t.is_open()) { ++ ABSL_LOG(ERROR) << "Failed to open file: " << filename; ++ return kTfLiteError; ++ } + std::string image_str((std::istreambuf_iterator(t)), + std::istreambuf_iterator()); + + tensorflow::png::DecodeContext context; -+ CHECK(CommonInitDecode(image_str, 3 /*RGB*/, 8 /*uint8*/, &context)); -+ char* image_buffer = new char[3 * context.width * context.height]; -+ CHECK(CommonFinishDecode(absl::bit_cast(image_buffer), -+ 3 * context.width /*stride*/, &context)); ++ // 0 channels means "detect from the input" rather than forcing 3, so ++ // greyscale and RGBA images load with the right number of components. ++ if (!tensorflow::png::CommonInitDecode(image_str, 0, 8 /*uint8*/, &context)) { ++ ABSL_LOG(ERROR) << "Failed to start decoding PNG image: " << filename; ++ return kTfLiteError; ++ } ++ const int image_size = context.channels * context.width * context.height; ++ // unsigned char rather than char: the signedness of plain char is ++ // implementation defined, and a signed buffer misreads every sample above ++ // 127 on iOS. ++ std::vector image_buffer(image_size); ++ if (!tensorflow::png::CommonFinishDecode( ++ absl::bit_cast(image_buffer.data()), ++ context.channels * context.width /*stride*/, &context)) { ++ ABSL_LOG(ERROR) << "Failed to decode PNG image: " << filename; ++ return kTfLiteError; ++ } + + image_data->width = context.width; + image_data->height = context.height; -+ std::vector* float_image = new std::vector(); -+ float_image->reserve(3 * context.width * context.height); -+ for (int i = 0; i < 3 * context.width * context.height; ++i) { -+ float_image->push_back(static_cast(image_buffer[i])); ++ image_data->data.clear(); ++ image_data->data.reserve(image_size); ++ for (int i = 0; i < image_size; ++i) { ++ image_data->data.push_back(static_cast(image_buffer[i])); + } -+ image_data->data.reset(float_image); ++ return kTfLiteOk; +} + // Central-cropping. - inline void Crop(ImageData* image_data, const CroppingParams& crop_params) { - int crop_height, crop_width; -@@ -288,6 +312,8 @@ TfLiteStatus ImagePreprocessingStage::Run() { - LoadImageRaw(image_path_, &image_data); - } else if (image_ext == ".jpg" || image_ext == ".jpeg") { - LoadImageJpeg(image_path_, &image_data); + inline TfLiteStatus Crop(ImageData* image_data, + const CroppingParams& crop_params) { +@@ -400,6 +443,10 @@ + if (LoadImageJpeg(*image_path_, &image_data) != kTfLiteOk) { + return kTfLiteError; + } + } else if (image_ext == ".png") { -+ LoadImagePng(image_path_, &image_data); ++ if (LoadImagePng(*image_path_, &image_data) != kTfLiteOk) { ++ return kTfLiteError; ++ } } else { LOG(ERROR) << "Extension " << image_ext << " is not supported"; return kTfLiteError; -diff --git a/tflite/tools/evaluation/stages/image_preprocessing_stage_test.cc b/tflite/tools/evaluation/stages/image_preprocessing_stage_test.cc -index 32105cbe7b4..05b83d0705b 100644 ---- a/tflite/tools/evaluation/stages/image_preprocessing_stage_test.cc -+++ b/tflite/tools/evaluation/stages/image_preprocessing_stage_test.cc -@@ -29,7 +29,7 @@ namespace { - constexpr char kImagePreprocessingStageName[] = "inception_preprocessing_stage"; - constexpr char kTestImage[] = - "tflite/tools/evaluation/stages/testdata/" -- "grace_hopper.jpg"; -+ "grace_hopper.png"; - constexpr int kImageDim = 224; - - TEST(ImagePreprocessingStage, NoParams) { --- -2.34.1 - diff --git a/flutter/third_party/png-with-number-of-channels-detected.patch b/flutter/third_party/png-with-number-of-channels-detected.patch deleted file mode 100644 index 47dea43c0..000000000 --- a/flutter/third_party/png-with-number-of-channels-detected.patch +++ /dev/null @@ -1,41 +0,0 @@ -From 63e353ca8a29515fdd12b0d1ce69800f144bb22b Mon Sep 17 00:00:00 2001 -From: Koan-Sin Tan -Date: Fri, 21 Apr 2023 06:25:00 +0800 -Subject: [PATCH] detect number of channels instead of forcing 3 (RGB) - ---- - .../evaluation/stages/image_preprocessing_stage.cc | 13 ++++++++----- - 1 file changed, 8 insertions(+), 5 deletions(-) - -diff --git a/tflite/tools/evaluation/stages/image_preprocessing_stage.cc b/tflite/tools/evaluation/stages/image_preprocessing_stage.cc -index a9750141b3d..47b5ed09a20 100644 ---- a/tflite/tools/evaluation/stages/image_preprocessing_stage.cc -+++ b/tflite/tools/evaluation/stages/image_preprocessing_stage.cc -@@ -120,16 +120,19 @@ inline void LoadImagePng(std::string* filename, ImageData* image_data) { - std::istreambuf_iterator()); - - tensorflow::png::DecodeContext context; -- CHECK(CommonInitDecode(image_str, 3 /*RGB*/, 8 /*uint8*/, &context)); -- char* image_buffer = new char[3 * context.width * context.height]; -+ // 0: channels is detected from the input -+ CHECK(CommonInitDecode(image_str, 0, 8 /*uint8*/, &context)); -+ char* image_buffer = -+ new char[context.channels * context.width * context.height]; - CHECK(CommonFinishDecode(absl::bit_cast(image_buffer), -- 3 * context.width /*stride*/, &context)); -+ context.channels * context.width /*stride*/, -+ &context)); - - image_data->width = context.width; - image_data->height = context.height; - std::vector* float_image = new std::vector(); -- float_image->reserve(3 * context.width * context.height); -- for (int i = 0; i < 3 * context.width * context.height; ++i) { -+ float_image->reserve(context.channels * context.width * context.height); -+ for (int i = 0; i < context.channels * context.width * context.height; ++i) { - float_image->push_back(static_cast(image_buffer[i])); - } - image_data->data.reset(float_image); ---- -2.34.1 - diff --git a/flutter/third_party/tensorflow-fix-file-opening-mode-for-Windows.patch b/flutter/third_party/tensorflow-fix-file-opening-mode-for-Windows.patch index 66f7a8c9a..14c922773 100644 --- a/flutter/third_party/tensorflow-fix-file-opening-mode-for-Windows.patch +++ b/flutter/third_party/tensorflow-fix-file-opening-mode-for-Windows.patch @@ -1,25 +1,18 @@ -From 572106fabc561a7f6338072fdab676c5bd2731c9 Mon Sep 17 00:00:00 2001 -From: Danil Uzlov -Date: Tue, 28 Sep 2021 10:28:53 +0700 Subject: [PATCH] fix file opening mode for Windows ---- - .../lite/tools/evaluation/stages/image_preprocessing_stage.cc | 2 +- - 1 file changed, 1 insertion(+), 1 deletion(-) +Originally by Danil Uzlov; regenerated for LiteRT 2.2.0. Applies on top of the +png patch, which is why the context shows LoadImagePng nearby. diff --git a/tflite/tools/evaluation/stages/image_preprocessing_stage.cc b/tflite/tools/evaluation/stages/image_preprocessing_stage.cc -index dd434a1c882..fe244b078ba 100644 --- a/tflite/tools/evaluation/stages/image_preprocessing_stage.cc +++ b/tflite/tools/evaluation/stages/image_preprocessing_stage.cc -@@ -83,7 +83,7 @@ inline void LoadImageRaw(std::string* filename, ImageData* image_data) { - // Loads the jpeg image. - inline void LoadImageJpeg(std::string* filename, ImageData* image_data) { +@@ -106,7 +106,8 @@ + inline TfLiteStatus LoadImageJpeg(absl::string_view filename, + ImageData* image_data) { // Reads image. -- std::ifstream t(*filename); -+ std::ifstream t(*filename, std::ios::binary); - std::string image_str((std::istreambuf_iterator(t)), - std::istreambuf_iterator()); - const int fsize = image_str.size(); --- -2.31.1.windows.1 - +- std::ifstream t(std::string(filename).c_str()); ++ std::ifstream t(std::string(filename).c_str(), ++ std::ios::in | std::ios::binary); + if (!t.is_open()) { + ABSL_LOG(ERROR) << "Failed to open file: " << filename; + return kTfLiteError; diff --git a/flutter/third_party/use_unsigned_char.patch b/flutter/third_party/use_unsigned_char.patch deleted file mode 100644 index 73aad1937..000000000 --- a/flutter/third_party/use_unsigned_char.patch +++ /dev/null @@ -1,24 +0,0 @@ -commit d8bdbe3eeacc607b773ef062aba04e64ec166aed -Author: freedom" Koan-Sin Tan -Date: Thu Jun 29 16:19:21 2023 +0800 - - signedness of char is undefined behavior - - use unsigned char instead of char to avoid causing problems - on iOS and others - -diff --git a/tflite/tools/evaluation/stages/image_preprocessing_stage.cc b/tflite/evaluation/stages/image_preprocessing_stage.cc -index 47b5ed09a20..224cc5da6c2 100644 ---- a/tflite/tools/evaluation/stages/image_preprocessing_stage.cc -+++ b/tflite/tools/evaluation/stages/image_preprocessing_stage.cc -@@ -122,8 +122,8 @@ inline void LoadImagePng(std::string* filename, ImageData* image_data) { - tensorflow::png::DecodeContext context; - // 0: channels is detected from the input - CHECK(CommonInitDecode(image_str, 0, 8 /*uint8*/, &context)); -- char* image_buffer = -- new char[context.channels * context.width * context.height]; -+ unsigned char* image_buffer = -+ new unsigned char[context.channels * context.width * context.height]; - CHECK(CommonFinishDecode(absl::bit_cast(image_buffer), - context.channels * context.width /*stride*/, - &context)); diff --git a/mobile_back_apple/cpp/backend_coreml/BUILD b/mobile_back_apple/cpp/backend_coreml/BUILD index 2caebeaa3..3882e20b1 100644 --- a/mobile_back_apple/cpp/backend_coreml/BUILD +++ b/mobile_back_apple/cpp/backend_coreml/BUILD @@ -38,8 +38,8 @@ apple_xcframework( "device": ["arm64"], }, minimum_os_versions = { - "ios": "13.1", - "macos": "13.1", + "ios": "15.0", + "macos": "15.0", }, deps = [ "//mobile_back_apple/cpp/backend_coreml:coreml_c", @@ -56,6 +56,12 @@ swift_library( srcs = [ "coreml_util.swift", ], + # Xcode 27 no longer defaults the language mode, so swiftc refuses to build + # without one. The source is mode-agnostic; 6 is what Xcode 27 would pick. + copts = [ + "-swift-version", + "6", + ], generates_header = True, deps = [ ":apple_frameworks", diff --git a/mobile_back_litert/README.md b/mobile_back_litert/README.md index 1f1bb248d..7058863c4 100644 --- a/mobile_back_litert/README.md +++ b/mobile_back_litert/README.md @@ -1,15 +1,18 @@ # Mobile backend LiteRT This backend runs all benchmarks on the -[LiteRT](https://github.com/google-ai-edge/LiteRT) 2.1.5 CompiledModel API: +[LiteRT](https://github.com/google-ai-edge/LiteRT) 2.2.0 CompiledModel API: the `llm-*` benchmarks on a dedicated LLM pipeline, `stable_diffusion` on a dedicated stable diffusion pipeline, and the vision/NLP benchmarks on a single-model pipeline. ## Overview -* Android arm64 only. The backend claims the `llm-*` benchmarks, - `stable_diffusion` and the vision/NLP benchmarks. +* Android arm64 and iOS arm64. Android claims all 13 benchmarks: the six + `llm-*` sizes, `stable_diffusion` and the vision/NLP benchmarks. iOS claims + nine of them -- everything except `llm-3b*` and `llm-8b*`, which cannot fit in + the iOS per-process memory limit. `llm-1b*` is claimed but does not fit on + every device either; see [iOS](#ios). * `claim_policy: CLAIM_SHARED`: lower-priority backends (e.g. the TFLite fallback) stay selectable next to LiteRT for every claimed benchmark. * `llm_pipeline.cc` drives a `litert::CompiledModel` with explicit `TensorBuffer`s @@ -43,6 +46,356 @@ Or build only the backend library: bazel build -c opt --config=android_arm64 //mobile_back_litert/cpp/backend_litert:liblitertbackend.so ``` +## iOS + +```bash +WITH_LITERT=1 make flutter/ios +``` + +* The GPU accelerator is the prebuilt `libLiteRtMetalAccelerator.dylib`, pinned + at 2.2.0 and downloaded by `litert_backend.mk`. It is embedded in + `Runner.app/Frameworks` next to the backend frameworks: LiteRT dlopens it from + the directory given by `kLiteRtEnvOptionTagRuntimeLibraryDir`, and iOS has no + dlopen search path to fall back on. +* It ships wrapped in `LiteRtMetalAccelerator.framework`, assembled by + `flutter/ios/ios.mk`. An iOS app bundle may only embed bundles, and a bare + Mach-O under `Frameworks/` is rejected by App Store Connect — that was + `ITMS-90426`, reported against Xcode Cloud build 265. The framework's + `CFBundleExecutable` is the dylib's original filename, so the path LiteRT + builds still resolves; `GetAppleRuntimeLibraryDir()` looks inside the + framework as well as beside the backend binary. +* Minimum iOS is 15.0: LiteRT's own floor (`LITERT_MIN_IOS_VERSION`), what the + prebuilt accelerator's Mach-O declares, and what the app already targets. The + sibling backend frameworks were raised from 13.1 to match, so every framework + the app embeds declares the same minimum. The accelerator framework's + `MinimumOSVersion` is read back out of the dylib by `ios.mk` rather than + written by hand: 2.2.0 raised the dylib from 14.0 to 15.0, and the 14.0 left + behind is what `ITMS-90208` rejected in build 269. +* The vision/NLP benchmarks offer both CPU and Metal and default to Metal, + which measured 1.7-3.0x faster than CPU across all six on an iPad mini. + `LiteRtGpuBackend` has no Metal enumerator: on Apple, Metal is selected by + `kLiteRtGpuBackendAutomatic` plus compile-time Metal support. +* `stable_diffusion` and the `llm-*` benchmarks each offer a Metal choice, and + **all three select it.** The LLM default follows the measurements below -- + Metal costs 72 MiB more at peak than CPU and decodes 3.0x faster -- and the + CI iPhone 16 Pro bears that out: 25.0 tok/s there against 9.27 on CPU. + `stable_diffusion` needed the model rewrite and the residency fix below + before it could take the choice at all; on the same device it now runs at + about 75 s per image against about 98 s on CPU. + The risk is worth stating plainly. An `EXC_RESOURCE` kill cannot be caught, + so if the budget does not stretch the app goes down rather than falling back. + The iPad mini crashes on Metal -- though it also fails `llm-1b` on **CPU**, + so it says nothing about this choice, and `llm-*` on Metal there is still + unmeasured while being claimed anyway. If a device run fails on memory, put + `delegate_selected` back to `CPU`; nothing else has to change. + +### Measuring this without a device + +The dev-utils CLI (`mobile_back_apple/dev-utils/Makefile`) runs a backend `.so` +on the host, which is far faster than an app build and is how the numbers below +were produced. macOS has no per-process jetsam cap, so a crash does not +reproduce there -- but `phys_footprint`, the counter `EXC_RESOURCE` compares +against, is reported at every phase, so "this delegate costs N MiB" transfers +straight to the device budget. + +```bash +bazel build -c opt --cxxopt=-std=c++17 --host_cxxopt=-std=c++17 \ + --macos_minimum_os=14.0 \ + //flutter/cpp/binary:main \ + //mobile_back_litert/cpp/backend_litert:liblitertbackend.so +# put the macos_arm64 accelerator next to the .so; the loose layout is what +# GetAppleRuntimeLibraryDir expects from a command-line harness +curl -fSL -o /libLiteRtMetalAccelerator.dylib \ + https://storage.googleapis.com/litert/binaries/2.2.0/macos_arm64/libLiteRtMetalAccelerator.dylib +bazel-bin/flutter/cpp/binary/main EXTERNAL llm-1b --mode=PerformanceOnly \ + --model_file= --lib_path=/liblitertbackend.so \ + --input_tfrecord=tinymmlu.tfrecord --sp_path=llama3_1b.spm.model +``` + +The CLI copies `delegate_selected` out of the settings verbatim and has no flag +to override it, so testing the Metal choice means flipping `delegate_selected` +in `litert_settings_apple.pbtxt` and rebuilding the `.so` (the settings are +compiled into it). + +### llm-1b on Metal, measured + +Apple GPU via the same Metal accelerator iOS uses, one MMLU query, MiB of +`phys_footprint`: + +| stage | CPU | Metal, sharing **on** | Metal, sharing **off** | +|---|---|---|---| +| after compile | 2094 | 2055 | 4497 | +| peak (prefill Run) | 3392 | 3464 | 5839 | +| time per output token | 49.10 ms | **16.28 ms** | -- | +| first token latency | 2.433 s | **1.229 s** | -- | +| TinyMMLU accuracy (100 samples) | 42.00% | 41.00% | -- | + +The accuracy line matters as much as the speed one: one sample in a hundred +separates them, which is what an fp16 GPU path against an int8/fp32 CPU path +should look like. Metal is not trading correctness for the 3x. + +Two things follow. **Metal costs what CPU costs** once constant-tensor sharing +is on -- 2055 against 2094 after compile, and 72 MiB more at peak -- so a +device that runs `llm-1b` on CPU has the headroom to run it on Metal. And +**Metal is 3.0x faster per output token** and 2.0x faster to first token, which +is the opposite of the Android GPU result (about half of CPU) that the earlier +"nothing is lost" reasoning leaned on. + +The third column is the bug. With sharing off the compile alone costs 4497 MiB +against a 3376 MB device cap, which is precisely the iPad mini crash: killed +inside `delegate_kernel.cc` "Initializing Metal-based API from graph", before +the model finished compiling. The log shows exactly two of those lines -- the +prefill and decode subgraphs -- and 4497 - 2055 = 2442 MiB is one extra copy of +the weights. That is what `EnableConstantTensorSharing` collapses. + +* Offering the LLM choice costs a download. `listResources()` in + `benchmark.dart` walks every `delegate_choice`, so the GPU export is fetched + whether or not Metal is ever selected: +1.25 GB on iOS, once, since `llm-1b` + and `llm-1b-instruct` name the same URL and the resource set dedupes. + Android already pays this. The `stable_diffusion` Metal choice costs nothing + extra -- it names the same four files as its CPU choice, and the + per-delegate model directory is symlinks into one shared download. +* The LLM Metal choice was withdrawn once and has now been reinstated with a + fix. What was measured on an iPad mini: the delegate is killed by + `EXC_RESOURCE` while still initialising (`delegate_kernel.cc`, "Initializing + Metal-based API from graph"), spending the whole ~3.0 GiB budget in about + 0.3 s, before the model finishes compiling and long before a prefill buffer + exists. The reading at the time was that the graph carries every prefill + signature and the delegate pays for all of them, against XNNPACK allocating + lazily per signature. + That is the symptom; the mechanism is `EnableConstantTensorSharing`. With it + off -- which is what the pipeline inherited from LiteRT-LM's Android options + -- constant tensors are *not* shared between subgraphs, so each signature + carries its own copy of weights that are 2072 MiB resident. Turning it on + routes them through LiteRT's mmap-backed `SharedMemoryManager`, and + `SetMadviseOriginalSharedTensors` lets the kernel drop the pages the layout + converter has already read. Clean file-backed pages do not count against + `phys_footprint`, which is what `EXC_RESOURCE` measures. Both are now on for + Apple only; Android's options are unchanged. + Note the same iPad mini also fails `llm-1b` on **CPU**, so it is not the + device to validate this on. An `EXC_RESOURCE` kill cannot be caught, so there + is no fallback to catch it: if the budget runs out the app goes down. Watch + the `[mem] llm: before GPU compile` and `[mem] llm: after GPU compile` lines + to see what the delegate actually costs. +* **`stable_diffusion` cannot use Metal with the published v5_0 exports, but + they can be rewritten so that it can** -- which is what the `*_litert_v2` + models the Metal choice downloads are. Each of the three v5_0 exports fails + to compile against the Metal accelerator: + + ```text + WARNING: Attempting to use a delegate that only supports static-sized tensors + with a graph that has dynamic-sized tensors + [probe] text_encoder on GPU: FAILED + [probe] diffusion on GPU: FAILED + [probe] decoder on GPU: FAILED + ``` + + This is not caused by the GPU options -- compiling with none of them set + fails identically -- and it is not partial delegation degrading to CPU, it is + `CompiledModel::Create` returning an error. + + The declared batch dimension looks like the culprit and is not. Each model + does declare `-1` for batch (`tokens [-1,77]`, `latent [-1,64,64,4]`, + `input_1 [-1,64,64,4]`), but rewriting `shape_signature` in all three so they + report **0 dynamic tensors** changes nothing. + + There are five blockers, not one, and `tools/sd_gpu/convert.py` fixes all + five; see that directory's README for the details and the measurements. Four + of them stop the delegate taking the graph at all and are listed here; the + fifth stops the memory option the delegate needs to fit on a phone, and is + the shader bug further down. + + 1. The graphs compute their shapes at run time (`SHAPE` -> + `REDUCE_PROD`/`GATHER`/`PACK`/`BROADCAST_ARGS` -> `RESHAPE`/`BROADCAST_TO`), + so a `RESHAPE` output stays dynamic however concrete the inputs are. + Since the pipeline only ever runs batch 1, those shapes are constants and + can be folded away. + 2. `BROADCAST_TO` is not implemented by the GPU delegate at all. + 3. The group-norm blocks work in rank 5, and the delegate refuses anything + above rank 4. + 4. `SUB` is declared at version 3 -- which only the rank-5 operands needed -- + against a delegate that supports version 2, which alone splits the graph + into partitions too small to delegate. + + With those four fixed the diffusion model and the decoder come out **fully** + GPU-accelerated and every output stays bit-identical. On an M-series Mac, ms + per invocation: + + | model | original CPU | rewritten CPU | rewritten Metal | + |---|---|---|---| + | text encoder | 12.4 | 13.1 | 14.1 | + | diffusion model | 1494.0 | 977.0 | 331.8 | + | decoder | 7702.3 | 2000.5 | 463.1 | + + This needs **LiteRT 2.2.0**, which is why this backend pins it. On 2.1.5 the + Metal backend emits invalid shader source for the int8 weights, + + ```text + newLibraryWithSource: program_source:30:30: error: use of undeclared identifier 'q0' + half4 weight_scale = half4(q0); + ``` + + with `AllowSrcQuantizedFcConvOps` both on and off, so it is a code-generation + bug rather than something the options control. The text encoder compiles on + 2.1.5; the diffusion model does not. The accelerator cannot be upgraded on + its own either: a 2.2.0 `libLiteRtMetalAccelerator.dylib` will not load into + a 2.1.5 runtime, nor the reverse. + + A near-identical bug survives into 2.2.0, and it is the reason for the fifth + rewrite: with **constant-tensor sharing on**, the diffusion model fails with + `use of undeclared identifier 'scale'`. Bisecting the options one at a time + shows sharing alone is the trigger. Sharing cannot simply be dropped -- it is + what makes the diffusion model fit on an iPhone at all -- so the models are + rewritten instead; the constant-tensor-sharing bullet below has the mechanism + and the numbers. `AllowSrcQuantizedFcConvOps` stays off either way: the + rewritten models are fully delegated without the quantized fc/conv path, and + dropping it also stops quantizing inputs to 8 bit, which + `litert_gpu_options.h` notes costs accuracy. + + End to end through this backend on a macOS host, one 20-step image: + + | models | delegate | per image | per step | + |---|---|---|---| + | published v5_0 | CPU | 43.84 s | ~2.1 s | + | rewritten | Metal | 18.8-19.4 s | ~0.82 s | + + Two runs, so the Metal figure is a range rather than a number. "Metal" here + means the pipeline compiled all three models on the delegate and never took + its CPU retry path; it is not a claim that no operator ran on CPU, and two + `GATHER`s in the text encoder do -- see `tools/sd_gpu/README.md`. + + For reference the rewrite is worth nothing on CPU on its own -- 43.22 s + against 43.84 s, one query each, which is noise -- so it only pays off + together with the delegate. + + When a GPU compile does fail the pipeline handles it rather than pretending: + everything built so far is released, the pages are handed back and the three + recompile on CPU. All three compile on one accelerator or none do, because + the benchmark reports a single accelerator name. On CPU an iPad mini measures + about 4.4-5.6 s per diffusion step, so roughly 100 s for 20 steps. +* **Constant tensor sharing is what makes the Metal path fit, and one shader + bug stood in the way of it.** The option decides whether the delegate + materialises the weights or keeps them stored and dequantises them in the + shader. Measured on a macOS host with `phys_footprint`, the counter + `EXC_RESOURCE` compares against: + + | model | file | sharing off | sharing on | + |---|---|---|---| + | text encoder (int8) | 118 MiB | 446 MiB | 118 MiB | + | diffusion model (int8) | 822 MiB | 4409 MiB | 1913 MiB | + | decoder (fp16) | 95 MiB | 1022 MiB | 217 MiB | + + About 2885 MiB is available to this benchmark on an iPhone 16 Pro, so with + sharing off the diffusion model does not fit on its own and no arrangement of + the pipeline rescues it. + + Sharing could not be turned on because the Metal backend generated a shader + that does not compile -- `use of undeclared identifier 'scale'`. It carries + two templates for dequantising int8 weights, one reading a per-axis scale + tensor and one taking a scalar, and it picks the scalar form for per-tensor + weights without declaring the arguments that form references. Single-op + models place the fault exactly: per-tensor int8 `FULLY_CONNECTED` fails, + while per-axis `FULLY_CONNECTED`, per-tensor `CONV_2D`, per-axis `CONV_2D` + and fp16 `CONV_2D` all compile. The accelerator is a prebuilt dylib, so the + model is the only side we control -- `tools/sd_gpu/convert.py` re-expresses + those weights as per-axis, repeating the one scale they already carry. The + weight bytes are untouched and the CPU output is bit-identical. +* **On the GPU the three models stay resident, because releasing one does not + give the memory back.** `ReleaseModel` drops the LiteRT objects and + `ReturnFreeMemoryToOS` empties libmalloc's free list, but the delegate's + weights live in Metal buffers libmalloc never owned. Measured on an iPhone 16 + Pro, in MiB still available before the limit: + + | stage | MiB left | | + |---|---|---| + | before compiling models | 2885 | | + | all three models compiled | 1517 | the three cost 1368 | + | query start, after releasing two of them | 1586 | only 69 recovered | + | diffusion model rebuilt | 72 | then `ActiveHard 3376 MB (fatal)` | + + Rebuilding stacks a second copy on top of the first. Compiled once and left + alone the three fit with the working set on top, so `set_phase` is simply not + installed on the GPU path — which also takes a 20-step image from 45.5 s to + 18.8 s on the macOS host, since nothing is recompiled per query. + + The CPU path keeps the phases: there the weights are ordinary allocations + that the release does return, and the three models plus a phase's working set + do not fit together. Android has the headroom, keeps everything resident, and + its throughput does not move. +* Releasing has to be symmetric, and this is easy to get wrong. Freeing only + the encoder and the diffusion model got the decode to pass, and then the + *second* query died: it rebuilt those two on top of the decoder's ~1.4 GiB + arena and had 541 MiB left for a phase that needs ~1405. +* Destroying a model is not sufficient on its own. `free()` does not + necessarily shrink the process footprint -- libmalloc keeps the pages on its + free list -- and that footprint is exactly what `EXC_RESOURCE` measures, so + the release can be invisible to the limit. `ReturnFreeMemoryToOS()` in + `apple_support.h` calls `malloc_zone_pressure_relief` to hand the pages back; + it is worth 2.3 GiB in the table above. This matters because LiteRT defers + `AllocateTensors` to the first `Run`, so a model's arena -- the largest + single allocation in the pipeline -- is created while it runs, on top of + whatever the allocator is still holding. +* When a memory question comes up here, measure it rather than reasoning from + model sizes -- that reasoning has been wrong more than once, most recently by + taking a model file's size for its resident cost. `LITERT_LOG_MEM("stage")` + logs `os_proc_available_memory()`, the budget `EXC_RESOURCE` actually + enforces, and `LITERT_LOG_NOTE(...)` logs a note beside it; both are no-ops + off Apple. They also go to `os_log`, because a BrowserStack device-log + artifact carries only `os_log` entries and drops native stderr entirely -- + without that a CI memory failure gives pass/fail and nothing to explain it. +* **Memory is the binding constraint for the LLM benchmarks, and it is decided + per device.** iOS kills a process that exceeds a per-process limit measured + at 3376 MB on an 8 GB device (`EXC_RESOURCE`). Measured for `llm-1b` on an + iPad mini, in MiB still available: + + | stage | MiB left | consumed | + |---|---|---| + | `backend_create` start | 2968 | app baseline 408 | + | model compiled | 896 | **2072** | + | decode buffers built | 894 | 2 | + | prefill buffers built | 653 | 241 | + | prefill inputs written | 461 | 192 | + | prefill `Run` | — | more than 461, killed | + + Two things there are worth keeping in mind. The weights cost **2072 MiB + resident against a 1229 MiB model file** -- XNNPACK repacks the q8 weights, + so file size is not a useful proxy. And this export publishes exactly **one** + prefill bucket, 1024, which the run above used on a 381-token prompt: there + is no smaller configuration to fall back to, so `llm-1b` simply does not fit + on that device. It does run on an iPhone 16 Pro (9.31 tok/s), which has more + headroom. +* **`llm-1b` is currently claimed on every iOS device anyway**, which is known + to be wrong for the iPad mini. A tested-device allowlist is the intended fix. + Gating on `os_proc_available_memory()` was tried and removed: that value is + not a device property. `mlperf_backend_matches_hardware` is called repeatedly, + and on one iPhone run it read anywhere from 3319 MiB down to 2600 MiB + depending on what had already run, so the benchmark list depended on when the + question was asked. A threshold picked to separate the two devices also came + within 9 MiB of excluding an iPhone 16 Pro, where the benchmark works. +* `llm-3b` and `llm-8b` are never offered, on any device: at the ratio above + their weights alone exceed the limit before a single buffer. `CLAIM_SHARED` + means the TFLite fallback still offers those four. +* The Apple prefill-bucket cap in `GetSuitablePrefillSignature` is a **no-op for + this export**, which publishes only the 1024 bucket. It is kept because it is + correct for any export that publishes several -- a bucket larger than the KV + cache can never be used, since a longer prompt is rejected outright -- but it + is not what makes anything fit here. Android passes `SIZE_MAX` and is + unaffected either way. +* Raising the ceiling is the other half of the approach, and + `flutter/ios/Runner/Runner.entitlements` now requests both + `com.apple.developer.kernel.increased-memory-limit` (raises the per-process + cap jetsam enforces) and + `com.apple.developer.kernel.extended-virtual-addressing` (lifts the + address-space limit the same allocations meet once the cap is up). + **They are inert until the matching capability is enabled for the App ID in + the developer portal**, and a provisioning profile that does not carry them + fails to sign every iOS backend -- so they are a separate commit, revertible + on its own if the signing setup is not ready. Neither helps `llm-8b`: 9.11 + GiB exceeds the RAM of the devices in question, which is why it is still not + claimed. +* There is no CoreML/ANE path: the LiteRT v2 API does not expose one yet + (upstream marks ANE "coming soon"). Use the Apple backend for CoreML. + ## Files * `cpp/backend_litert/litert_c.cc` — MLPerf backend C API implementation and @@ -55,8 +408,13 @@ bazel build -c opt --config=android_arm64 //mobile_back_litert/cpp/backend_liter diffusion pipeline: text encoder, diffusion loop and decoder. It runs on the CPU accelerator; the shipped models are `dynamic_int8` (text encoder, diffusion) and `dynamic_fp16` (decoder) exports aimed at CPU/XNNPACK. -* `cpp/backend_litert/backend_settings/litert_settings_android.pbtxt` — benchmark settings (models, delegates). -* `litert_backend.mk` — make variables and the GPU accelerator download. +* `cpp/backend_litert/litert_env.h` — builds the `litert::Environment` shared by + the pipelines. +* `cpp/backend_litert/apple_support.h` — locates the accelerator inside the app bundle on iOS. +* `cpp/backend_litert/backend_settings/litert_settings_android.pbtxt` and + `litert_settings_apple.pbtxt` — benchmark settings (models, delegates). +* `cpp/backend_litert/ios/BUILD` — the `liblitertbackend.xcframework` bundle. +* `litert_backend.mk` — make variables and the GPU accelerator downloads. Models and tokenizers are downloaded from `mobile.mlcommons-storage.org` as defined in the settings file. diff --git a/mobile_back_litert/cpp/backend_dummy/BUILD b/mobile_back_litert/cpp/backend_dummy/BUILD deleted file mode 100644 index 0e675bfd1..000000000 --- a/mobile_back_litert/cpp/backend_dummy/BUILD +++ /dev/null @@ -1,28 +0,0 @@ -# Copyright 2022 The MLPerf Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# ============================================================================== - -package( - default_visibility = ["//visibility:public"], - licenses = ["notice"], # Apache 2.0 -) - -cc_library( - name = "dummy_backend", - srcs = ["dummy_backend.cc"], - deps = [ - "//flutter/cpp/c:headers", - ], - alwayslink = 1, -) diff --git a/mobile_back_litert/cpp/backend_dummy/dummy_backend.cc b/mobile_back_litert/cpp/backend_dummy/dummy_backend.cc deleted file mode 100644 index 4d60cfea7..000000000 --- a/mobile_back_litert/cpp/backend_dummy/dummy_backend.cc +++ /dev/null @@ -1,71 +0,0 @@ - -#include "flutter/cpp/c/backend_c.h" - -bool mlperf_backend_matches_hardware(const char** not_allowed_message, - const char** settings, - const mlperf_device_info_t* device_info) { - return false; -} - -mlperf_backend_ptr_t mlperf_backend_create( - const char* model_path, mlperf_backend_configuration_t* configs, - const char* native_lib_path) { - return nullptr; -} - -const char* mlperf_backend_vendor_name(mlperf_backend_ptr_t backend_ptr) { - return ""; -} - -const char* mlperf_backend_accelerator_name(mlperf_backend_ptr_t backend_ptr) { - return ""; -} - -const char* mlperf_backend_name(mlperf_backend_ptr_t backend_ptr) { return ""; } - -void mlperf_backend_delete(mlperf_backend_ptr_t backend_ptr) {} - -mlperf_status_t mlperf_backend_issue_query(mlperf_backend_ptr_t backend_ptr, - ft_callback callback, - void* context) { - return MLPERF_FAILURE; -} - -mlperf_status_t mlperf_backend_flush_queries(mlperf_backend_ptr_t backend_ptr) { - return MLPERF_FAILURE; -} - -int32_t mlperf_backend_get_input_count(mlperf_backend_ptr_t backend_ptr) { - return 0; -} - -mlperf_data_t mlperf_backend_get_input_type(mlperf_backend_ptr_t backend_ptr, - int32_t i) { - mlperf_data_t result; - result.type = mlperf_data_t::Float32; - result.size = 0; - return result; -} -mlperf_status_t mlperf_backend_set_input(mlperf_backend_ptr_t backend_ptr, - int32_t batchIndex, int32_t i, - void* data) { - return MLPERF_FAILURE; -} - -int32_t mlperf_backend_get_output_count(mlperf_backend_ptr_t backend_ptr) { - return 0; -} - -mlperf_data_t mlperf_backend_get_output_type(mlperf_backend_ptr_t backend_ptr, - int32_t i) { - mlperf_data_t result; - result.type = mlperf_data_t::Float32; - result.size = 0; - return result; -} - -mlperf_status_t mlperf_backend_get_output(mlperf_backend_ptr_t backend_ptr, - uint32_t batchIndex, int32_t i, - void** data) { - return MLPERF_FAILURE; -} diff --git a/mobile_back_litert/cpp/backend_dummy/ios/BUILD b/mobile_back_litert/cpp/backend_dummy/ios/BUILD deleted file mode 100644 index 9acc33eff..000000000 --- a/mobile_back_litert/cpp/backend_dummy/ios/BUILD +++ /dev/null @@ -1,37 +0,0 @@ -# Copyright 2022 The MLPerf Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# ============================================================================== - -load("@build_bazel_rules_apple//apple:apple.bzl", "apple_xcframework") - -# This target is required to have the same name -# as the true litert backend. -# Otherwise xcode will fail to build the app. -apple_xcframework( - name = "liblitertbackend", - bundle_id = "com.mlcommons.inference.backend-litert-dummy", - exported_symbols_lists = ["//flutter/cpp/c:exported_symbols.lds"], - infoplists = ["//flutter/cpp/flutter:BackendBridgeInfo.plist"], - ios = { - "simulator": ["x86_64"], - "device": ["arm64"], - }, - minimum_os_versions = { - "ios": "13.1", - "macos": "13.1", - }, - deps = [ - "//mobile_back_litert/cpp/backend_dummy:dummy_backend", - ], -) diff --git a/mobile_back_litert/cpp/backend_litert/BUILD b/mobile_back_litert/cpp/backend_litert/BUILD index 1c0df6a26..43311b37e 100644 --- a/mobile_back_litert/cpp/backend_litert/BUILD +++ b/mobile_back_litert/cpp/backend_litert/BUILD @@ -24,10 +24,17 @@ package( licenses = ["notice"], # Apache 2.0 ) +exports_files( + [ + "exported_symbols_litert.lds", + ], +) + pbtxt2header( name = "litert_settings", srcs = [ "backend_settings/litert_settings_android.pbtxt", + "backend_settings/litert_settings_apple.pbtxt", ], ) @@ -43,8 +50,11 @@ cc_library( "stable_diffusion_pipeline.cc", ], hdrs = [ + "apple_support.h", "embedding_utils.h", + "litert_env.h", "litert_settings_android.h", + "litert_settings_apple.h", "llm_pipeline.h", "pipeline.h", "sd_utils.h", @@ -62,6 +72,10 @@ cc_library( ], "//conditions:default": [], }), + # No Apple-only deps: unlike the legacy interpreter backends this one talks + # to the v2 CompiledModel API, and Metal support is compiled into the litert + # runtime on __APPLE__ and provided at runtime by the dlopened + # libLiteRtMetalAccelerator.dylib. deps = [ ":litert_settings", "//flutter/cpp:utils", @@ -81,6 +95,8 @@ cc_library( "@litert//litert/cc:litert_compiled_model", "@litert//litert/cc:litert_element_type", "@litert//litert/cc:litert_environment", + "@litert//litert/cc:litert_environment_options", + "@litert//litert/cc:litert_expected", "@litert//litert/cc:litert_model", "@litert//litert/cc:litert_options", "@litert//litert/cc:litert_tensor_buffer", @@ -90,8 +106,14 @@ cc_library( tflite_jni_binary( name = "liblitertbackend.so", - # Not //flutter/cpp/c:version_script.lds — this backend also exports the - # LiteRt* symbols for the dlopened GPU accelerator library. + # macOS uses exported_symbols, other platforms use linkscript -- passing + # only one of them silently produces a library that exports nothing on the + # other. Without exported_symbols this target links on macOS but yields a + # 179K stub with no mlperf_backend_* symbols, so the dev-utils CLI in + # mobile_back_apple cannot load it. Neither file is the //flutter/cpp/c one + # the sibling backends use: this backend also exports the LiteRt* symbols + # for the dlopened GPU accelerator library. + exported_symbols = "exported_symbols_litert.lds", linkscript = "version_script.lds", deps = [ ":litert_c", diff --git a/mobile_back_litert/cpp/backend_litert/apple_support.h b/mobile_back_litert/cpp/backend_litert/apple_support.h new file mode 100644 index 000000000..265ad61c3 --- /dev/null +++ b/mobile_back_litert/cpp/backend_litert/apple_support.h @@ -0,0 +1,165 @@ +/* Copyright 2026 The MLPerf Authors. All Rights Reserved. + +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + +Unless required by applicable law or agreed to in writing, software +distributed under the License is distributed on an "AS IS" BASIS, +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +See the License for the specific language governing permissions and +limitations under the License. +==============================================================================*/ +#ifndef LITERT_APPLE_SUPPORT_H_ +#define LITERT_APPLE_SUPPORT_H_ + +#if defined(__APPLE__) + +#include +#include +#include +#include +#include + +#include + +#if TARGET_OS_IPHONE +#include +#else +// os_proc_available_memory() is explicitly unavailable on macOS: there is no +// per-process high-watermark limit there to be available against. The dev-utils +// CLI (mobile_back_apple/dev-utils) runs this backend on the host, so the file +// still has to compile; report phys_footprint instead, which is the same +// quantity EXC_RESOURCE measures on device and is what makes host runs +// comparable to device ones. +#include +#include +#endif // TARGET_OS_IPHONE + +#include "absl/log/log.h" + +// LiteRT dlopens its GPU accelerator (libLiteRtMetalAccelerator.dylib on Apple) +// by joining the kLiteRtEnvOptionTagRuntimeLibraryDir environment option with +// the library filename. That option is mandatory here: iOS has no dlopen +// directory search, and the app passes an empty native_lib_path to +// mlperf_backend_create (device_info.dart returns '' on iOS), so the directory +// has to be recovered from this binary's own location instead. +namespace litert_apple { + +// How many bytes this process can still allocate before iOS kills it. This is +// the budget EXC_RESOURCE enforces (3376 MB on an 8 GB device), and it is the +// only thing that decides which benchmarks this backend can claim, so log it +// around the phases that allocate rather than reasoning about it from model +// sizes. Cheap: a counter read, not a scan. +inline void LogAvailableMemory(const char* stage) { +#if TARGET_OS_IPHONE + const size_t available = os_proc_available_memory(); + // 0 means the call is unavailable (it needs an app context), not that the + // process is out of memory -- do not report that as an imminent kill. + if (available == 0) return; + const size_t mib = available / (1024 * 1024); + const char* const unit = "MiB left before the iOS limit"; +#else + // macOS: report what the process is holding rather than what is left, since + // nothing is enforcing a ceiling here. phys_footprint is the same counter + // EXC_RESOURCE compares against on device, so a host measurement of "this + // delegate costs N MiB" transfers directly to the device budget. + task_vm_info_data_t info = {}; + mach_msg_type_number_t count = TASK_VM_INFO_COUNT; + if (task_info(mach_task_self(), TASK_VM_INFO, + reinterpret_cast(&info), &count) != KERN_SUCCESS) { + return; + } + const size_t mib = static_cast(info.phys_footprint) / (1024 * 1024); + const char* const unit = "MiB phys_footprint"; +#endif // TARGET_OS_IPHONE + LOG(INFO) << "[mem] " << stage << ": " << mib << " " << unit; + // Also to os_log, which is the only one of the two that reaches a + // BrowserStack device-log artifact: that capture carries os_log entries + // (Dart's print arrives that way) but no native stderr, so without this a CI + // memory failure gives pass/fail and nothing to diagnose it with. + os_log(OS_LOG_DEFAULT, "[mem] %{public}s: %zu %{public}s", stage, mib, unit); +} + +// Hand pages the allocator is holding back to the OS. +// +// Destroying a compiled model frees its weights and arenas, but free() does +// not necessarily shrink the process footprint: libmalloc keeps the pages in +// its free list and stays ready to reuse them. phys_footprint still counts +// them, and phys_footprint is exactly what EXC_RESOURCE measures -- so a +// release that looks correct can leave the budget unchanged. A null zone means +// every zone, and a goal of 0 means reclaim as much as possible. +inline void ReturnFreeMemoryToOS() { malloc_zone_pressure_relief(nullptr, 0); } + +// Log a diagnostic line to both sinks, for the same reason LogAvailableMemory +// does: absl goes to native stderr, which a local `flutter run` shows but a +// BrowserStack device-log artifact drops entirely. +inline void LogNote(const std::string& text) { + LOG(INFO) << text; + os_log(OS_LOG_DEFAULT, "%{public}s", text.c_str()); +} + +inline bool FileExists(const std::string& path) { + struct stat info; + return stat(path.c_str(), &info) == 0; +} + +// Returns the parent directory of `path`, or an empty string if there is none. +inline std::string DirName(const std::string& path) { + const size_t slash = path.rfind('/'); + if (slash == std::string::npos) return ""; + if (slash == 0) return "/"; + return path.substr(0, slash); +} + +} // namespace litert_apple + +// Returns the directory holding libLiteRtMetalAccelerator.dylib inside the app +// bundle, or an empty string when it cannot be found (the caller then runs on +// CPU). dladdr gives this binary's path, e.g. +// .../MyApp.app/Frameworks/liblitertbackend.framework/liblitertbackend, so the +// search starts there and walks up to .../MyApp.app/Frameworks. +// +// The accelerator ships wrapped in LiteRtMetalAccelerator.framework, because an +// iOS app bundle may only embed bundles and App Store Connect rejects a bare +// Mach-O under Frameworks/. The framework's CFBundleExecutable is the dylib's +// original filename, so the path this returns still joins with that name the +// way LiteRT expects. The loose layout is still accepted at each level: it is +// what a non-bundle host (a macOS command-line harness) produces. +inline std::string GetAppleRuntimeLibraryDir() { + static constexpr char kAcceleratorName[] = "libLiteRtMetalAccelerator.dylib"; + static constexpr char kAcceleratorFramework[] = + "LiteRtMetalAccelerator.framework"; + + Dl_info info; + if (dladdr(reinterpret_cast(&GetAppleRuntimeLibraryDir), + &info) == 0 || + info.dli_fname == nullptr) { + LOG(WARNING) << "dladdr failed to locate the LiteRT backend binary; the " + "Metal accelerator will not be loaded"; + return ""; + } + + std::string dir = litert_apple::DirName(std::string(info.dli_fname)); + for (int level = 0; level < 2 && !dir.empty(); ++level) { + const std::string candidates[] = {dir, dir + "/" + kAcceleratorFramework}; + for (const std::string& candidate : candidates) { + if (litert_apple::FileExists(candidate + "/" + kAcceleratorName)) { + LOG(INFO) << "LiteRT runtime library dir: " << candidate; + return candidate; + } + } + dir = litert_apple::DirName(dir); + } + + LOG(WARNING) << kAcceleratorName << " not found near " << info.dli_fname + << "; the Metal accelerator is unavailable and the pipeline " + "falls back to CPU"; + return ""; +} + +#endif // defined(__APPLE__) + +#endif // LITERT_APPLE_SUPPORT_H_ diff --git a/mobile_back_litert/cpp/backend_litert/backend_settings/litert_settings_apple.pbtxt b/mobile_back_litert/cpp/backend_litert/backend_settings/litert_settings_apple.pbtxt new file mode 100644 index 000000000..7947c2f0e --- /dev/null +++ b/mobile_back_litert/cpp/backend_litert/backend_settings/litert_settings_apple.pbtxt @@ -0,0 +1,517 @@ +# proto-file: flutter/cpp/proto/backend_setting.proto +# proto-message: BackendSetting + +# Apple/iOS settings for the LiteRT backend. Every benchmark runs on the LiteRT +# CompiledModel API: llm-* on the LLM pipeline, stable_diffusion on the stable +# diffusion pipeline, the rest on the single-model pipeline. The Metal choice +# is served by the prebuilt +# libLiteRtMetalAccelerator.dylib that ios.mk drops next to the backend +# frameworks in the app bundle's Frameworks directory and LiteRT dlopens at +# environment setup; the CPU choice runs on XNNPACK and keeps +# the int8 exports benchmarkable. No ANE choice is offered: the CompiledModel +# NPU path needs vendor SDKs and AOT-compiled models. Lower-priority backends +# (the TFLite fallback) stay selectable next to it, so users can compare +# implementations per benchmark. +# +# The vision/NLP benchmarks default to Metal, measured on an iPad mini: Metal +# ran image_classification_v2 at 77.3 QPS against 44.9 on CPU, and completed +# object_detection, image_segmentation_v2, natural_language_processing and +# image_classification_offline_v2 with valid results. super_resolution follows +# the same pattern but has not been measured; if its GPU compile ever fails the +# pipeline falls back to CPU on its own. +# +# The llm-* benchmarks are offered on every device, which is known not to hold +# on all of them -- see the note above the llm-1b entry. They and +# stable_diffusion now offer a Metal choice as well, neither of them selected: +# both are there to be measured, not to be defaults. llm-3b and llm-8b are +# never offered; see the note at the bottom. +claim_policy: CLAIM_SHARED + +common_setting { + id: "num_threads" + name: "Number of threads" + value { + value: "4" + name: "4 threads" + } +} + +benchmark_setting { + benchmark_id: "image_classification_v2" + framework: "LiteRT" + delegate_choice: { + delegate_name: "CPU" + accelerator_name: "cpu" + accelerator_desc: "CPU" + model_file: { + model_path: "https://mobile.mlcommons-storage.org/app-resources/models/v4_0/tflite/MobileNetV4-Conv-Large-int8-ptq.tflite" + model_checksum: "590a7a88640a18d28b16b6f571cdfc93" + } + } + delegate_choice: { + delegate_name: "Metal" + accelerator_name: "gpu" + accelerator_desc: "GPU" + model_file: { + model_path: "https://mobile.mlcommons-storage.org/app-resources/models/v4_0/tflite/MobileNetV4-Conv-Large-fp32.tflite" + model_checksum: "b57cc2a027607c3b36873a15ace84acb" + } + } + delegate_selected: "Metal" +} + +benchmark_setting { + benchmark_id: "object_detection" + framework: "LiteRT" + delegate_choice: { + delegate_name: "CPU" + accelerator_name: "cpu" + accelerator_desc: "CPU" + model_file: { + model_path: "https://mobile.mlcommons-storage.org/app-resources/models/v1_0/mobiledet_qat.tflite" + model_checksum: "6c7af49d97a2b2488222d94936d2dc18" + } + } + delegate_choice: { + delegate_name: "Metal" + accelerator_name: "gpu" + accelerator_desc: "GPU" + model_file: { + model_path: "https://mobile.mlcommons-storage.org/app-resources/models/v1_0/mobiledet.tflite" + model_checksum: "566ceb72a4c7c8926fe4ac8eededb5bf" + } + } + delegate_selected: "Metal" +} + +benchmark_setting { + benchmark_id: "image_segmentation_v2" + framework: "LiteRT" + delegate_choice: { + delegate_name: "CPU" + accelerator_name: "cpu" + accelerator_desc: "CPU" + model_file: { + model_path: "https://mobile.mlcommons-storage.org/app-resources/models/v2_0/mobile_segmenter_r4_quant_argmax_uint8.tflite" + model_checksum: "b7a7620b8b818d64305b51ab796bfb1d" + } + } + delegate_choice: { + delegate_name: "Metal" + accelerator_name: "gpu" + accelerator_desc: "GPU" + model_file: { + model_path: "https://mobile.mlcommons-storage.org/app-resources/models/v2_0/mobile_segmenter_r4_argmax_f32.tflite" + model_checksum: "b3a5d3c2e5756431a471ed5211c344a9" + } + } + delegate_selected: "Metal" +} + +benchmark_setting { + benchmark_id: "natural_language_processing" + framework: "LiteRT" + delegate_choice: { + delegate_name: "CPU" + accelerator_name: "cpu" + accelerator_desc: "CPU" + model_file: { + model_path: "https://mobile.mlcommons-storage.org/app-resources/models/v0_7/mobilebert_int8_384_nnapi.tflite" + model_checksum: "3944a2dee04a5f8a5fd016ac34c4d390" + } + } + delegate_choice: { + delegate_name: "Metal" + accelerator_name: "gpu" + accelerator_desc: "GPU" + model_file: { + model_path: "https://mobile.mlcommons-storage.org/app-resources/models/v0_7/mobilebert_float_384_gpu.tflite" + model_checksum: "36a953d07a8c6f2d3e05b22e87cec95b" + } + } + delegate_selected: "Metal" +} + +benchmark_setting { + benchmark_id: "super_resolution" + framework: "LiteRT" + delegate_choice: { + delegate_name: "CPU" + accelerator_name: "cpu" + accelerator_desc: "CPU" + model_file: { + model_path: "https://mobile.mlcommons-storage.org/app-resources/models/v3_0/edsr_f32b5_full_qint8.tflite" + model_checksum: "18ce6df0e4603f4b4ee5d04193708d9c" + } + } + delegate_choice: { + delegate_name: "Metal" + accelerator_name: "gpu" + accelerator_desc: "GPU" + model_file: { + model_path: "https://mobile.mlcommons-storage.org/app-resources/models/v3_0/edsr_f32b5_fp32.tflite" + model_checksum: "672240427c1f3dc33baf2facacd9631f" + } + } + delegate_selected: "Metal" +} + +benchmark_setting { + benchmark_id: "image_classification_offline_v2" + framework: "LiteRT" + delegate_choice: { + delegate_name: "CPU" + accelerator_name: "cpu" + accelerator_desc: "CPU" + model_file: { + model_path: "https://mobile.mlcommons-storage.org/app-resources/models/v4_0/tflite/MobileNetV4-Conv-Large-int8-ptq.tflite" + model_checksum: "590a7a88640a18d28b16b6f571cdfc93" + } + batch_size: 2 + } + delegate_choice: { + delegate_name: "Metal" + accelerator_name: "gpu" + accelerator_desc: "GPU" + model_file: { + model_path: "https://mobile.mlcommons-storage.org/app-resources/models/v4_0/tflite/MobileNetV4-Conv-Large-fp32.tflite" + model_checksum: "b57cc2a027607c3b36873a15ace84acb" + } + batch_size: 2 + } + delegate_selected: "Metal" + # One CompiledModel with the batch dimension resized to 2, matching the + # Android setting: concurrent Run calls from multiple shards deadlock on the + # shared LiteRT GPU context (observed on Pixel 10 Pro), and the Metal + # accelerator has not been exercised with multiple shards either. + custom_setting { + id: "shards_num" + value: "1" + } +} + +benchmark_setting { + benchmark_id: "stable_diffusion" + framework: "LiteRT" + # MEASURED: Metal works for this benchmark, but only with the exports + # produced by tools/sd_gpu/convert.py and only on LiteRT 2.2.0 or newer. + # + # The published v5_0 models do not compile on the Metal accelerator at all: + # CompiledModel::Create fails with "a delegate that only supports + # static-sized tensors with a graph that has dynamic-sized tensors", for the + # text encoder, diffusion model and decoder alike. Setting no GPU options at + # all fails identically, so it is not the option set, and rewriting + # shape_signature so the models report 0 dynamic tensors does not help + # either: the graphs compute their shapes at RUN TIME (SHAPE -> + # REDUCE_PROD/GATHER/PACK/BROADCAST_ARGS -> RESHAPE/BROADCAST_TO), which + # makes the outputs dynamic whatever the inputs say. + # + # Folding those shapes away is only the first of four blockers. BROADCAST_TO + # is not implemented by the delegate; the group-norm blocks work in rank 5 + # against a rank-4 limit; and SUB is declared at version 3, which only the + # rank-5 operands needed, against a delegate that supports version 2 -- that + # last one alone splits the graph into partitions too small to delegate. + # tools/sd_gpu/README.md carries the details and the measurements. + # + # The Metal choice below therefore names its own model files. They are the + # same networks: the converter runs the original and the rewrite on the same + # inputs and refuses to write anything that is not bit-identical. + # + # Constant-tensor sharing is ON, and this benchmark does not fit without it. + # The option decides whether the delegate materialises the weights or keeps + # them stored and dequantises them in the shader: compiling the diffusion + # model on Metal costs 4409 MiB with it off against 1913 MiB with it on, + # measured on a macOS host with phys_footprint, the counter EXC_RESOURCE + # compares against. Roughly 2885 MiB is available to this benchmark on an + # iPhone 16 Pro, so with sharing off no arrangement of the pipeline fits. + # + # Sharing was off until the converter learned its fifth rewrite. With it on + # the Metal backend used to emit invalid shader source ("use of undeclared + # identifier 'scale'"): it picks a scalar weight-dequant template for + # per-tensor int8 weights but never declares the arguments that template + # references. Single-op models place the fault exactly -- per-tensor int8 + # FULLY_CONNECTED fails, per-axis FULLY_CONNECTED and every CONV_2D form + # compile -- so tools/sd_gpu/convert.py re-expresses those weights as + # per-axis, repeating the one scale they already carry. The weight bytes are + # untouched and the CPU output is bit-identical. + # + # AllowSrcQuantizedFcConvOps stays off. litert_gpu_options.h says sharing + # "must be true to use this", so it is available now, but the rewritten + # models are fully delegated without it and the header notes it quantizes the + # input tensors to 8 bit, which costs accuracy. + # + # Measured end to end through this backend on a macOS host, one 20-step + # image: 43.84 s on CPU with the published models against 18.8-19.4 s on + # Metal with these, over two runs. Per denoising step that is about 2.1 s + # against 0.82 s. "On Metal" means all three models compiled on the delegate + # and the pipeline never took its CPU retry path -- not that no operator ran + # on CPU; two GATHERs in the text encoder still do. + # + # The three model files total about 1.04 GiB, which is a download size, not a + # peak footprint. Measured on the macOS host: 118 MiB for the text encoder, + # 1913 MiB for the diffusion model and 217 MiB for the decoder, 2015 MiB with + # all three compiled at once, and about 2.3-2.7 GiB at the peak. On an + # iPhone 16 Pro the three cost 1368 MiB of the 2885 available. They are + # compiled once and left resident on the GPU: releasing one does not return + # its Metal buffers, so rebuilding per phase only stacks a second copy and + # crosses the limit. See SDBackendData::set_phase. + delegate_choice: { + delegate_name: "CPU" + accelerator_name: "cpu" + accelerator_desc: "CPU" + model_file: { + model_path: "https://mobile.mlcommons-storage.org/app-resources/models/v5_0/tflite/sd_decoder_dynamic_fp16.tflite" + model_checksum: "165b70a01643e70a23e5e54a949be306" + } + model_file: { + model_path: "https://mobile.mlcommons-storage.org/app-resources/models/v5_0/tflite/sd_diffusion_model_dynamic_int8.tflite" + model_checksum: "ccfd761a2f8186c3669948515d40a880" + } + model_file: { + model_path: "https://mobile.mlcommons-storage.org/app-resources/models/v5_0/tflite/sd_text_encoder_dynamic_int8.tflite" + model_checksum: "b64effb0360f9ea49a117cdaf8a2fbdc" + } + model_file: { + model_path: "https://mobile.mlcommons-storage.org/app-resources/models/v5_0/tflite/timestep_embeddings_data.bin.ts" + model_checksum: "798b772155a69de5df44b304327bb3cc" + } + } + delegate_choice: { + delegate_name: "Metal" + accelerator_name: "gpu" + accelerator_desc: "GPU" + # The GPU-ready exports, produced by tools/sd_gpu/convert.py from the v5_0 + # models above. They are the same networks -- the converter refuses to + # write a model whose output is not bit-identical -- rewritten so the + # delegate will take them. The CPU choice keeps the published files. + model_file: { + model_path: "https://storage.googleapis.com/mlperf-mobile-public/litert/sd_decoder_litert.tflite" + model_checksum: "8aa94e17f9394958e0c71c653ab1140f" + } + model_file: { + model_path: "https://storage.googleapis.com/mlperf-mobile-public/litert/sd_diffusion_model_litert_v2.tflite" + model_checksum: "7bd563de431b8253ea482fdbdd513a47" + } + model_file: { + model_path: "https://storage.googleapis.com/mlperf-mobile-public/litert/sd_text_encoder_litert_v2.tflite" + model_checksum: "be6310001157f50bed34750db2c88486" + } + model_file: { + model_path: "https://mobile.mlcommons-storage.org/app-resources/models/v5_0/tflite/timestep_embeddings_data.bin.ts" + model_checksum: "798b772155a69de5df44b304327bb3cc" + } + custom_setting { + id: "text_encoder_filename" + value: "sd_text_encoder_litert_v2.tflite" + } + custom_setting { + id: "diffusion_model_filename" + value: "sd_diffusion_model_litert_v2.tflite" + } + custom_setting { + id: "decoder_filename" + value: "sd_decoder_litert.tflite" + } + } + delegate_selected: "Metal" + custom_setting { + id: "pipeline" + value: "StableDiffusionPipeline" + } + custom_setting { + id: "text_encoder_filename" + value: "sd_text_encoder_dynamic_int8.tflite" + } + custom_setting { + id: "diffusion_model_filename" + value: "sd_diffusion_model_dynamic_int8.tflite" + } + custom_setting { + id: "decoder_filename" + value: "sd_decoder_dynamic_fp16.tflite" + } + custom_setting { + id: "timestep_embeddings_filename" + value: "timestep_embeddings_data.bin.ts" + } +} + +# The llm-* benchmarks are claimed on every iOS device, and that is known to be +# wrong for some of them: llm-1b is killed by EXC_RESOURCE on an iPad mini. +# Measured there (limit 3376 MB), MiB still available: +# +# backend_create start 2968 (app baseline 408) +# model compiled 896 (the q8 weights cost 2072 resident, not the +# 1229 of the .tflite file -- XNNPACK repacks) +# decode buffers built 894 +# prefill buffers built 653 +# prefill inputs written 461 +# prefill Run -> killed, the arena needs more than 461 +# +# The weights alone take 2072 of 3376 MB and there is no smaller configuration +# to fall back on: this export publishes exactly one prefill bucket (1024), used +# above on a 381-token prompt. It does run on an iPhone 16 Pro (9.31 tok/s). +# +# Gating this on os_proc_available_memory() was tried and removed. The value is +# not a device property: mlperf_backend_matches_hardware is called repeatedly +# and it read anywhere from 3319 MiB down to 2600 MiB on one iPhone run +# depending on what had already run, so the benchmark list depended on when the +# question was asked. A tested-device allowlist is the intended replacement. +# +# The Metal choice below was withdrawn once (see the git history) because the +# delegate was killed by EXC_RESOURCE while still initialising. The cause was +# GPU constant-tensor sharing being off, which gives each subgraph its own copy +# of the weights; the pipeline now turns it on for Apple. +# +# MEASURED on a macOS host through mobile_back_apple/dev-utils, MiB of +# phys_footprint (the counter EXC_RESOURCE compares against), one MMLU query: +# +# CPU Metal on Metal off +# after compile 2094 2055 4497 +# peak (prefill Run) 3392 3464 5839 +# time per output token 49.10ms 16.28ms -- +# TinyMMLU accuracy 42.00% 41.00% -- +# +# One sample in a hundred separates the accuracies, which is what an fp16 GPU +# path against an int8/fp32 CPU path should look like -- the 3x is not being +# bought with correctness. +# +# So Metal costs what CPU costs (+72 MiB at peak) and decodes 3.0x faster, +# while the unfixed path needed 4497 MiB to compile against a 3376 MB device +# cap -- which is the crash, exactly. +# +# Metal is now what is selected, so the CI iPhone 16 Pro exercises it. That is +# the device the host numbers predict should hold: llm-1b already passes there +# on CPU, and Metal costs 72 MiB more at peak while decoding 3x faster. +# +# It is still the riskier of the two choices, and the risk is worth stating +# plainly: an EXC_RESOURCE kill cannot be caught, so if the budget does not +# stretch the app goes down rather than falling back to CPU, and the one iOS +# device this was ever tried on (an iPad mini) crashed -- though that device +# also fails llm-1b on CPU, so it proves nothing about this one. If the CI +# device run fails on memory, put this line back to "CPU"; everything else in +# this file stays as it is. +# +# Offering it is not free: listResources() in benchmark.dart walks every +# delegate_choice, so the GPU export is downloaded whether or not Metal is +# ever selected. That is +1.25 GB on iOS, once -- llm-1b and llm-1b-instruct +# name the same URL and the resource set dedupes. Android already pays it. +# The stable_diffusion Metal choice below costs nothing extra by contrast: it +# names the same four files as its CPU choice, and the per-delegate model +# directory is symlinks into one shared download. + +benchmark_setting { + benchmark_id: "llm-1b" + framework: "LiteRT" + delegate_choice: { + delegate_name: "CPU" + accelerator_name: "cpu" + accelerator_desc: "CPU" + model_file: { + model_path: "https://mobile.mlcommons-storage.org/app-resources/models/v6_0/tflite/llama_q8_ekv3072.tflite" + model_checksum: "c618e9bbb89ce52eedcab4f61b2dc3a4" + } + model_file: { + model_path: "https://mobile.mlcommons-storage.org/app-resources/datasets/v6_0/llama3_1b.spm.model" + model_checksum: "2ad260fc18b965ce16006d76c9327082" + } + } + # The Metal choice uses the dedicated GPU export, exactly as on Android; the + # delegate-level model_filename overrides the benchmark-level one. It is + # offered but not selected: see the note above this benchmark. + delegate_choice: { + delegate_name: "Metal" + accelerator_name: "gpu" + accelerator_desc: "GPU" + model_file: { + model_path: "https://storage.googleapis.com/mlperf-mobile-public/litert/llama_q8_ekv3072_litert.tflite" + model_checksum: "fcabc03f8361f7081e13e89df84a0ce5" + } + model_file: { + model_path: "https://mobile.mlcommons-storage.org/app-resources/datasets/v6_0/llama3_1b.spm.model" + model_checksum: "2ad260fc18b965ce16006d76c9327082" + } + custom_setting { + id: "model_filename" + value: "llama_q8_ekv3072_litert.tflite" + } + } + delegate_selected: "Metal" + custom_setting { + id: "pipeline" + value: "LLMPipeline" + } + custom_setting { + id: "model_filename" + value: "llama_q8_ekv3072.tflite" + } + custom_setting { + id: "tokenizer_filename" + value: "llama3_1b.spm.model" + } +} + +benchmark_setting { + benchmark_id: "llm-1b-instruct" + framework: "LiteRT" + delegate_choice: { + delegate_name: "CPU" + accelerator_name: "cpu" + accelerator_desc: "CPU" + model_file: { + model_path: "https://mobile.mlcommons-storage.org/app-resources/models/v6_0/tflite/llama_q8_ekv3072.tflite" + model_checksum: "c618e9bbb89ce52eedcab4f61b2dc3a4" + } + model_file: { + model_path: "https://mobile.mlcommons-storage.org/app-resources/datasets/v6_0/llama3_1b.spm.model" + model_checksum: "2ad260fc18b965ce16006d76c9327082" + } + } + # The Metal choice uses the dedicated GPU export, exactly as on Android; the + # delegate-level model_filename overrides the benchmark-level one. It is + # offered but not selected: see the note above this benchmark. + delegate_choice: { + delegate_name: "Metal" + accelerator_name: "gpu" + accelerator_desc: "GPU" + model_file: { + model_path: "https://storage.googleapis.com/mlperf-mobile-public/litert/llama_q8_ekv3072_litert.tflite" + model_checksum: "fcabc03f8361f7081e13e89df84a0ce5" + } + model_file: { + model_path: "https://mobile.mlcommons-storage.org/app-resources/datasets/v6_0/llama3_1b.spm.model" + model_checksum: "2ad260fc18b965ce16006d76c9327082" + } + custom_setting { + id: "model_filename" + value: "llama_q8_ekv3072_litert.tflite" + } + } + delegate_selected: "Metal" + custom_setting { + id: "pipeline" + value: "LLMPipeline" + } + custom_setting { + id: "model_filename" + value: "llama_q8_ekv3072.tflite" + } + custom_setting { + id: "tokenizer_filename" + value: "llama3_1b.spm.model" + } +} + +# llm-3b and llm-8b are NEVER offered on iOS, on any device, which is why they +# are absent from the gated file as well. iOS kills a process past a per-process +# limit measured at 3376 MB on an 8 GB device (EXC_RESOURCE), and their weights +# alone exceed it. llm-1b was measured at 2072 MiB resident against a 1229 MiB +# model file -- XNNPACK repacks the q8 weights -- so scaling that ratio: +# +# llm-3b: 3.05 GiB of model file, over 5 GiB resident before any buffer. +# llm-8b: 7.55 GiB of model file, more than the device has in total. +# +# No prefill tuning reaches that, so claiming them would only offer the user a +# benchmark that kills the app. claim_policy is CLAIM_SHARED, so the TFLite +# fallback still offers these benchmarks. Android keeps all six sizes; it has +# far more headroom. diff --git a/mobile_back_litert/cpp/backend_litert/exported_symbols_litert.lds b/mobile_back_litert/cpp/backend_litert/exported_symbols_litert.lds new file mode 100644 index 000000000..66ee28d91 --- /dev/null +++ b/mobile_back_litert/cpp/backend_litert/exported_symbols_litert.lds @@ -0,0 +1,11 @@ +# Export the symbols in backend_c.h, plus the LiteRT C API. +# +# The LiteRt* line mirrors the Android version_script.lds. Inspecting the +# prebuilt 2.1.5 libLiteRtMetalAccelerator.dylib shows it is self-contained -- +# it defines the LiteRt* symbols it needs and imports none of them from its +# loader -- so this is insurance for a future accelerator build that does +# import them, not a present-day requirement. It costs binary size (the LiteRT +# surface can no longer be dead-stripped); drop it once a device run confirms +# the Metal accelerator still registers without it. +_mlperf_backend_* +_LiteRt* diff --git a/mobile_back_litert/cpp/backend_litert/ios/BUILD b/mobile_back_litert/cpp/backend_litert/ios/BUILD new file mode 100644 index 000000000..92fc096fc --- /dev/null +++ b/mobile_back_litert/cpp/backend_litert/ios/BUILD @@ -0,0 +1,47 @@ +# Copyright 2026 The MLPerf Authors. All Rights Reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# ============================================================================== + +load("@build_bazel_rules_apple//apple:apple.bzl", "apple_xcframework") + +apple_xcframework( + name = "liblitertbackend", + bundle_id = "com.mlcommons.inference.backend-litert", + # Not //flutter/cpp/c:exported_symbols.lds — this backend also exports the + # LiteRt* symbols, mirroring the Android version script. See that file for + # why they are kept even though the 2.1.5 Metal accelerator does not need + # them. + exported_symbols_lists = ["//mobile_back_litert/cpp/backend_litert:exported_symbols_litert.lds"], + infoplists = ["//flutter/cpp/flutter:BackendBridgeInfo.plist"], + ios = { + "simulator": [ + "x86_64", + # cpuinfo doesn't support simulator on ARM-based macs + # "arm64", + ], + "device": ["arm64"], + }, + # litert/cc/BUILD sets LITERT_MIN_IOS_VERSION = "15.0", so anything linking + # the LiteRT runtime has to declare at least that, and the prebuilt Metal + # accelerator is built for 15.0 as well. The sibling backends were raised to + # match rather than left at 13.1, so every framework in the app declares one + # minimum -- the same one the app itself targets. + minimum_os_versions = { + "ios": "15.0", + "macos": "15.0", + }, + deps = [ + "//mobile_back_litert/cpp/backend_litert:litert_c", + ], +) diff --git a/mobile_back_litert/cpp/backend_litert/litert_c.cc b/mobile_back_litert/cpp/backend_litert/litert_c.cc index 6a74fadda..2898978c5 100644 --- a/mobile_back_litert/cpp/backend_litert/litert_c.cc +++ b/mobile_back_litert/cpp/backend_litert/litert_c.cc @@ -13,26 +13,38 @@ limitations under the License. #include #include "absl/log/log.h" -#include "litert_settings_android.h" #include "llm_pipeline.h" #include "single_model_pipeline.h" #include "stable_diffusion_pipeline.h" +#if defined(__APPLE__) +#include "litert_settings_apple.h" +#elif defined(__ANDROID__) +#include "litert_settings_android.h" +#endif + #ifdef __cplusplus extern "C" { #endif // __cplusplus std::unique_ptr pipeline; -// This backend is Android-only. The llm-* benchmarks run on the LiteRT -// CompiledModel LLM pipeline, stable_diffusion runs on the CompiledModel -// stable diffusion pipeline (CPU), and every other benchmark runs on the -// CompiledModel single-model pipeline (GPU accelerator with CPU fallback). +// This backend supports Android and Apple (iOS). The llm-* benchmarks run on +// the LiteRT CompiledModel LLM pipeline, stable_diffusion runs on the +// CompiledModel stable diffusion pipeline (CPU on both platforms), and +// every other benchmark runs on the CompiledModel single-model pipeline (GPU +// accelerator with CPU fallback). The GPU accelerator is the Metal one on Apple +// and the OpenCL/GL one on Android; LiteRT picks it from the compile-time +// platform support, so each platform only differs in its settings file. bool mlperf_backend_matches_hardware(const char **not_allowed_message, const char **settings, const mlperf_device_info_t *device_info) { *not_allowed_message = nullptr; -#ifdef __ANDROID__ +#if defined(__APPLE__) + *settings = litert_settings_apple.c_str(); + LOG(INFO) << "LiteRT backend matches hardware"; + return true; +#elif defined(__ANDROID__) // Samsung Galaxy M32 (SM-M326B) does not have enough memory to run LLM // benchmarks, so don't offer this backend there. if (device_info->model != nullptr && diff --git a/mobile_back_litert/cpp/backend_litert/litert_env.h b/mobile_back_litert/cpp/backend_litert/litert_env.h new file mode 100644 index 000000000..4f60749e4 --- /dev/null +++ b/mobile_back_litert/cpp/backend_litert/litert_env.h @@ -0,0 +1,81 @@ +/* Copyright 2026 The MLPerf Authors. All Rights Reserved. + +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + +Unless required by applicable law or agreed to in writing, software +distributed under the License is distributed on an "AS IS" BASIS, +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +See the License for the specific language governing permissions and +limitations under the License. +==============================================================================*/ +#ifndef LITERT_ENV_H_ +#define LITERT_ENV_H_ + +#include +#include +#include + +#include "absl/log/log.h" +#include "litert/cc/litert_environment.h" +#include "litert/cc/litert_environment_options.h" +#include "litert/cc/litert_expected.h" + +#if defined(__APPLE__) +#include "apple_support.h" +#endif + +// Log the remaining process memory budget at a phase boundary. Only Apple +// enforces a per-process limit tight enough to matter here, so this is a no-op +// everywhere else and costs nothing. +#if defined(__APPLE__) +#define LITERT_LOG_MEM(stage) ::litert_apple::LogAvailableMemory(stage) +#else +#define LITERT_LOG_MEM(stage) ((void)0) +#endif + +// Log a stream-style diagnostic. Same line on every platform; on Apple it also +// goes to os_log, which is the only sink a device-log artifact captures. +// +// LITERT_LOG_NOTE("chose bucket " << seq << " for " << n << " tokens"); +#if defined(__APPLE__) +#define LITERT_LOG_NOTE(stream_expr) \ + do { \ + std::ostringstream litert_note_stream; \ + litert_note_stream << stream_expr; \ + ::litert_apple::LogNote(litert_note_stream.str()); \ + } while (0) +#else +#define LITERT_LOG_NOTE(stream_expr) LOG(INFO) << stream_expr +#endif + +// Creates the LiteRT environment the compiled models are built in. Both +// pipelines share it so the platform-conditional setup stays in one place. +// +// On Apple the GPU (Metal) accelerator is dlopened from the runtime library +// directory, which has to be passed explicitly; see apple_support.h. Everywhere +// else LiteRT uses its default library search, so no options are needed. +inline litert::Expected CreateLiteRtEnvironment() { +#if defined(__APPLE__) + const std::string runtime_library_dir = GetAppleRuntimeLibraryDir(); + if (!runtime_library_dir.empty()) { + const std::vector env_options = { + {litert::EnvironmentOptions::Tag::kRuntimeLibraryDir, + runtime_library_dir.c_str()}}; + auto env = litert::Environment::Create(litert::EnvironmentOptions( + litert::Span( + env_options.data(), env_options.size()))); + if (env) return env; + // The directory is only a hint for the accelerator loader, so never fail + // the backend over it: without it the compile falls back to CPU. + LOG(WARNING) << "Environment::Create with a runtime library dir failed; " + << "retrying without it"; + } +#endif + return litert::Environment::Create({}); +} + +#endif // LITERT_ENV_H_ diff --git a/mobile_back_litert/cpp/backend_litert/litert_settings_apple.h b/mobile_back_litert/cpp/backend_litert/litert_settings_apple.h new file mode 100644 index 000000000..2b7f6a218 --- /dev/null +++ b/mobile_back_litert/cpp/backend_litert/litert_settings_apple.h @@ -0,0 +1,22 @@ +/* Copyright 2026 The MLPerf Authors. All Rights Reserved. + +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + +Unless required by applicable law or agreed to in writing, software +distributed under the License is distributed on an "AS IS" BASIS, +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +See the License for the specific language governing permissions and +limitations under the License. +==============================================================================*/ +#ifndef LITERT_SETTINGS_APPLE_H +#define LITERT_SETTINGS_APPLE_H + +#include "mobile_back_litert/cpp/backend_litert/backend_settings/litert_settings_apple.pbtxt.h" + +const std::string litert_settings_apple = litert_settings_apple_pbtxt; + +#endif diff --git a/mobile_back_litert/cpp/backend_litert/llm_pipeline.cc b/mobile_back_litert/cpp/backend_litert/llm_pipeline.cc index 526e6cb12..f49a6bea3 100644 --- a/mobile_back_litert/cpp/backend_litert/llm_pipeline.cc +++ b/mobile_back_litert/cpp/backend_litert/llm_pipeline.cc @@ -29,9 +29,12 @@ limitations under the License. #include "litert/cc/litert_compiled_model.h" #include "litert/cc/litert_element_type.h" #include "litert/cc/litert_environment.h" +#include "litert/cc/litert_environment_options.h" +#include "litert/cc/litert_expected.h" #include "litert/cc/litert_options.h" #include "litert/cc/litert_tensor_buffer.h" #include "litert/cc/options/litert_gpu_options.h" +#include "litert_env.h" #ifdef __cplusplus extern "C" { @@ -49,11 +52,53 @@ static std::unordered_map make_index_map( return map; } +// Report what a buffer set actually costs, and which single tensor dominates +// it. The Apple memory decisions for this pipeline were made from arithmetic +// over model files and tensor shapes, and that arithmetic was wrong once +// already (the KV cache is four resident sets, not two). These are the +// measured numbers. +static void LogBufferSet(const char* label, + const std::vector& bufs, + const std::vector& names) { + size_t total = 0; + size_t largest = 0; + std::string largest_name = "?"; + for (size_t i = 0; i < bufs.size(); ++i) { + auto packed = bufs[i].PackedSize(); + if (!packed) continue; + total += *packed; + if (*packed > largest) { + largest = *packed; + largest_name = i < names.size() ? std::string(names[i]) : "?"; + } + } + LITERT_LOG_NOTE("[mem] llm " << label << ": " << (total >> 20) << " MiB over " + << bufs.size() << " tensors, largest '" + << largest_name << "' " << (largest >> 20) + << " MiB"); +} + // Destroy the backend pointer and its data. void LLMPipeline::backend_delete(mlperf_backend_ptr_t backend_ptr) { LLMBackendData* backend_data = (LLMBackendData*)backend_ptr; if (backend_data) delete backend_data; backendExists = false; + // Reported after the delete so a benchmark that follows this one can be told + // apart from one that inherits memory this pipeline never gave back. + LITERT_LOG_MEM("llm: backend deleted (before reclaim)"); +#if defined(__APPLE__) + // Destroying the model is not enough to make a release visible to the limit: + // free() leaves the pages on libmalloc's free list and phys_footprint -- + // which is what EXC_RESOURCE measures -- still counts them. The stable + // diffusion pipeline reclaims for this reason and measured 2.3 GiB from it. + // + // This only reaches host allocations. It cannot release GPU resources the + // runtime never freed, so it is an adjunct here, not the fix for the Metal + // teardown leak -- see patches/custom_buffer_teardown.patch for that. The + // two log lines around it are what tell the difference apart on device. + litert_apple::ReturnFreeMemoryToOS(); + LITERT_LOG_MEM("llm: backend deleted (after reclaim)"); +#endif // defined(__APPLE__) } // Create a new backend and return the pointer to it. @@ -66,6 +111,8 @@ mlperf_backend_ptr_t LLMPipeline::backend_create( return nullptr; } + LITERT_LOG_MEM("llm: backend_create start"); + LLMBackendData* backend_data = new LLMBackendData(); std::string model_dir = std::string(model_path); @@ -120,12 +167,14 @@ mlperf_backend_ptr_t LLMPipeline::backend_create( return nullptr; } } + LITERT_LOG_MEM("llm: model compiled"); if (!BuildDecodeBuffers(*backend_data)) { LOG(ERROR) << "Failed to allocate decode buffers"; backend_delete(backend_data); return nullptr; } + LITERT_LOG_MEM("llm: decode buffers built"); backendExists = true; return backend_data; @@ -192,10 +241,15 @@ mlperf_status_t LLMPipeline::backend_issue_first_token_query( backend_data->prefill_input_bufs[backend_data->prefill_mask_idx]); } + // The observed EXC_RESOURCE lands in a memmove around here. LiteRT defers + // AllocateTensors to the first Run, so this call is where the prefill + // signature's arena appears -- not at buffer-creation time. + LITERT_LOG_MEM("llm: prefill inputs written, about to Run"); MINIMAL_CHECK(backend_data->model->Run( backend_data->current_prefill_sig_idx, absl::MakeSpan(backend_data->prefill_input_bufs), absl::MakeSpan(backend_data->prefill_output_bufs))); + LITERT_LOG_MEM("llm: prefill Run done"); // Move the prefill KV state into the decode buffers. TransferKV(backend_data->num_kv_layers, backend_data->prefill_output_map, @@ -346,8 +400,39 @@ mlperf_status_t LLMPipeline::backend_set_input(mlperf_backend_ptr_t backend_ptr, backend_data->kv_v_float_counts, backend_data->decode_input_map, backend_data->decode_input_bufs); size_t effective_prefill_token_size = backend_data->prompt_tokens.size() - 1; + // The prefill logits buffer is [1, bucket, vocab] fp32, so every step up in + // bucket size doubles it -- for the 1B export that is 1.05 GiB at 2048 and + // 2.10 GiB at 4096. On Apple that overshoots the per-process memory limit + // (measured at 3376 MB on an 8 GB device) once the ~1.2 GiB model is in, and + // the app is killed with EXC_RESOURCE. A bucket bigger than the KV cache can + // never be used anyway: the check below rejects any prompt longer than + // kv_cache_max_size, so tokens past it would have nowhere to go. Capping + // there keeps the largest usable bucket and drops the wasted half. Android + // keeps the uncapped choice so its validated throughput does not move. +#if defined(__APPLE__) + const size_t max_useful_seq_size = + static_cast(backend_data->kv_cache_max_size); +#else + const size_t max_useful_seq_size = std::numeric_limits::max(); +#endif size_t sig_idx = GetSuitablePrefillSignature(backend_data->prefill_sigs, - effective_prefill_token_size); + effective_prefill_token_size, + max_useful_seq_size); + { + // The one fact that decides the prefill logits buffer, and the one this + // pipeline never reported: which bucket the cap actually selected. + size_t chosen_seq = 0; + for (const auto& [idx, seq] : backend_data->prefill_sigs) { + if (idx == sig_idx) chosen_seq = seq; + } + const bool capped = + max_useful_seq_size != std::numeric_limits::max(); + LITERT_LOG_NOTE( + "[mem] llm bucket choice: prompt " + << (effective_prefill_token_size + 1) << " tokens, cap " + << (capped ? std::to_string(max_useful_seq_size) : std::string("none")) + << " -> bucket " << chosen_seq); + } if (!BuildPrefillBuffers(*backend_data, sig_idx)) return MLPERF_FAILURE; if ((int)(effective_prefill_token_size + 1) > backend_data->kv_cache_max_size) { @@ -400,7 +485,7 @@ void LLMPipeline::backend_release_buffer(void* p) { ::operator delete(p); } bool LLMPipeline::BuildCompiledModel(LLMBackendData& data, const char* model_path, bool use_gpu) { - auto env = litert::Environment::Create({}); + auto env = CreateLiteRtEnvironment(); if (!env) { LOG(ERROR) << "Environment::Create failed"; return false; @@ -435,9 +520,38 @@ bool LLMPipeline::BuildCompiledModel(LLMBackendData& data, gpu_options->AddExternalTensorPattern("kv_cache_"); // Prefill and decode must each land in a single delegate partition. gpu_options->SetHintFullyDelegatedToSingleDelegate(true); - gpu_options->SetMadviseOriginalSharedTensors(false); gpu_options->SetConvertWeightsOnGpu(false); +#if defined(__APPLE__) + // Apple diverges from LiteRT-LM here, and it is the difference between + // running and being killed. Metal materialises every signature in the + // graph before the first Run, and with constant-tensor sharing off each + // subgraph carries its own copy of the transformer weights -- 2072 MiB + // resident, per subgraph. On an iPad mini that ate the whole ~3.0 GiB + // budget in about 0.3 s during "Initializing Metal-based API from graph" + // and the process died in __bzero under EXC_RESOURCE, before the model + // finished compiling and long before any prefill buffer existed. + // + // Sharing routes those constants through LiteRT's SharedMemoryManager, + // which is mmap-backed, so the copies collapse to one mapping; madvising + // the originals then lets the kernel drop the pages the layout converter + // has already read. phys_footprint is what EXC_RESOURCE measures, and + // clean file-backed pages do not count against it, so both flags have to + // be on to move the number that kills us. The documented cost is slower + // tensor binding through the external-tensor APIs, which is the right + // trade against not running at all. + // + // It is also a documented prerequisite of the option set just below: + // litert_gpu_options.h says enable_constant_tensors_sharing "must be true + // to use" AllowSrcQuantizedFcConvOps. The non-Apple branch inherits that + // contradiction from LiteRT-LM and is left alone here -- Android's GPU LLM + // path is measured and working, and this change is scoped to iOS -- but it + // is worth knowing that the quantized-op option is probably inert there. + gpu_options->EnableConstantTensorSharing(true); + gpu_options->SetMadviseOriginalSharedTensors(true); +#else + gpu_options->SetMadviseOriginalSharedTensors(false); gpu_options->EnableConstantTensorSharing(false); +#endif // defined(__APPLE__) gpu_options->EnableAllowSrcQuantizedFcConvOps(true); // KV cache is swapped, so GPU bindings repeat every 2 steps. gpu_options->SetNumStepsOfCommandBufferPreparations(2); @@ -450,6 +564,10 @@ bool LLMPipeline::BuildCompiledModel(LLMBackendData& data, LOG(ERROR) << "GetGpuOptions failed; GPU compile may crash"; } + // The GPU compile is where this pipeline has historically been killed + // rather than returning an error, so record the budget on the way in: a + // process that dies inside Create leaves this as the last line written. + LITERT_LOG_MEM("llm: before GPU compile"); std::string transformer_path = model_path; auto model = litert::CompiledModel::Create(*data.env, transformer_path, *options); @@ -457,6 +575,7 @@ bool LLMPipeline::BuildCompiledModel(LLMBackendData& data, LOG(ERROR) << "CompiledModel::Create failed: " << transformer_path; return false; } + LITERT_LOG_MEM("llm: after GPU compile"); data.model = std::make_unique(std::move(*model)); auto decode = data.model->GetSignatureIndex("decode"); @@ -491,6 +610,16 @@ bool LLMPipeline::BuildCompiledModel(LLMBackendData& data, std::sort(data.prefill_sigs.begin(), data.prefill_sigs.end(), [](const auto& a, const auto& b) { return a.second < b.second; }); + { + std::ostringstream buckets; + for (const auto& [sig_idx, seq] : data.prefill_sigs) { + if (buckets.tellp() > 0) buckets << ", "; + buckets << seq; + } + LITERT_LOG_NOTE( + "[mem] llm prefill buckets exported by the model: " << buckets.str()); + } + return true; } @@ -841,6 +970,13 @@ bool LLMPipeline::BuildDecodeBuffers(LLMBackendData& data) { auto buffer_size = logits_metadata->BufferSize(); data.vocab_size = static_cast(*buffer_size / sizeof(float)); + + LITERT_LOG_NOTE("[mem] llm geometry: kv_cache_max_size=" + << data.kv_cache_max_size << " num_kv_layers=" + << data.num_kv_layers << " vocab_size=" << data.vocab_size + << " (decode logits " << (*buffer_size >> 20) << " MiB)"); + LogBufferSet("decode inputs", data.decode_input_bufs, input_names); + LogBufferSet("decode outputs", data.decode_output_bufs, output_names); return true; } @@ -917,22 +1053,44 @@ bool LLMPipeline::BuildPrefillBuffers(LLMBackendData& data, } } + LITERT_LOG_NOTE("[mem] llm prefill buffers built for bucket " + << data.prefill_seq_size); + LogBufferSet("prefill inputs", data.prefill_input_bufs, input_names); + LogBufferSet("prefill outputs", data.prefill_output_bufs, output_names); + LITERT_LOG_MEM("llm: prefill buffers built"); + data.current_prefill_sig_idx = prefill_sig_idx; return true; } size_t LLMPipeline::GetSuitablePrefillSignature( const std::vector>& prefill_sigs, - size_t num_input_tokens) const { - size_t best = prefill_sigs.back().first; + size_t num_input_tokens, size_t max_useful_seq_size) const { + // The smallest bucket that fits the prompt without exceeding the cap. size_t delta = std::numeric_limits::max(); + size_t best = 0; for (const auto& [sig_idx, seq_size] : prefill_sigs) { + if (seq_size > max_useful_seq_size) continue; if (seq_size >= num_input_tokens && seq_size - num_input_tokens < delta) { delta = seq_size - num_input_tokens; best = sig_idx; } } - return best; + if (delta != std::numeric_limits::max()) return best; + + // Nothing within the cap is large enough for the prompt, so take the largest + // bucket that is still within it: the caller prefills that many tokens and + // decodes the remainder one at a time, which is slower but correct. + // prefill_sigs is sorted by sequence size, so the last match is the largest. + bool found = false; + for (const auto& [sig_idx, seq_size] : prefill_sigs) { + if (seq_size <= max_useful_seq_size) { + best = sig_idx; + found = true; + } + } + // The cap is below every bucket; the smallest one is the best we can do. + return found ? best : prefill_sigs.front().first; } void LLMPipeline::TransferKV( diff --git a/mobile_back_litert/cpp/backend_litert/llm_pipeline.h b/mobile_back_litert/cpp/backend_litert/llm_pipeline.h index c06fa764b..718e89c27 100644 --- a/mobile_back_litert/cpp/backend_litert/llm_pipeline.h +++ b/mobile_back_litert/cpp/backend_litert/llm_pipeline.h @@ -57,8 +57,16 @@ struct LLMBackendData { const char* vendor = "Google"; const char* accelerator = "CPU"; - std::unique_ptr model; + // LiteRT requires the environment to outlive the compiled model built in it + // ("the provided environment must outlive the compiled model and any + // executions running on it" -- litert_compiled_model.h). The destructor + // below is what enforces that here: it clears the buffers, then the model, + // then the environment, and a destructor body runs before its members are + // destroyed. Declaration order is kept consistent with it anyway -- and with + // the ordering SDBackendData documents -- so the two cannot drift if that + // destructor is ever simplified away. std::unique_ptr env; + std::unique_ptr model; std::vector> prefill_sigs; // (sig_idx, seq_size) std::unique_ptr embedder; @@ -208,9 +216,12 @@ class LLMPipeline : public Pipeline { bool RunEmbedders(LLMBackendData& data, const std::vector& tokens, bool prefill); + // Pick the prefill signature to run the prompt through. Buckets larger than + // max_useful_seq_size are skipped when a smaller one exists; pass SIZE_MAX + // to consider every bucket. size_t GetSuitablePrefillSignature( const std::vector>& prefill_sigs, - size_t num_input_tokens) const; + size_t num_input_tokens, size_t max_useful_seq_size) const; // Move each layer's KV from the prefill outputs into the decode inputs. void TransferKV( int num_layers, diff --git a/mobile_back_litert/cpp/backend_litert/single_model_pipeline.cc b/mobile_back_litert/cpp/backend_litert/single_model_pipeline.cc index ce332c350..3a6d5a3e4 100644 --- a/mobile_back_litert/cpp/backend_litert/single_model_pipeline.cc +++ b/mobile_back_litert/cpp/backend_litert/single_model_pipeline.cc @@ -27,20 +27,32 @@ limitations under the License. #include #endif +#if defined(__APPLE__) +#include +#endif + #include "absl/log/log.h" #include "flutter/cpp/c/type.h" #include "litert/cc/litert_compiled_model.h" #include "litert/cc/litert_element_type.h" #include "litert/cc/litert_environment.h" +#include "litert/cc/litert_environment_options.h" +#include "litert/cc/litert_expected.h" #include "litert/cc/litert_model.h" #include "litert/cc/litert_options.h" #include "litert/cc/litert_tensor_buffer.h" +#include "litert_env.h" #include "thread_pool.h" namespace { constexpr char kDelegateCpu[] = "CPU"; constexpr char kDelegateGpu[] = "GPU"; +#if defined(__APPLE__) +// The Apple settings name the GPU choice after the accelerator that serves it, +// the same way the TFLite backend's iOS settings do. +constexpr char kDelegateMetal[] = "Metal"; +#endif // The vision/NLP models expose a single (default) signature. constexpr size_t kSignatureIndex = 0; @@ -276,6 +288,23 @@ bool BuildShards(LiteRTBackendData *backend_data, const char *model_path, return fail(); } } + // Propagate the new shapes before creating the buffers. A resize only + // marks the signature as needing allocation; LiteRT defers AllocateTensors + // to Run, so the output tensors still report their pre-resize sizes here. + // CreateOutputBuffers sizes a CPU buffer from exactly that value, and Run + // then registers it as a TFLite custom allocation and reallocates, which + // fails with "Custom allocation is too small for tensor idx" once the real + // shapes land. Asking for the output layouts with update_allocation forces + // the allocation first. The GPU path is unaffected (its requirements come + // from the accelerator, not from tensor->bytes), and Run allocates again + // anyway, so this is safe for both. + auto layouts = backend_data->shards[k].GetOutputTensorLayouts( + kSignatureIndex, /*update_allocation=*/true); + if (!layouts) { + LOG(ERROR) << "Failed to update the tensor allocation for shard " << k + << ": " << layouts.Error().Message(); + return fail(); + } auto input_bufs = backend_data->shards[k].CreateInputBuffers(); auto output_bufs = backend_data->shards[k].CreateOutputBuffers(); if (!input_bufs || !output_bufs) { @@ -357,7 +386,7 @@ mlperf_backend_ptr_t SingleModelPipeline::backend_create( backend_data->executer = std::make_unique(backend_data->shards_num); - auto env = litert::Environment::Create({}); + auto env = CreateLiteRtEnvironment(); if (!env) { LOG(ERROR) << "Environment::Create failed"; backend_delete(backend_data); @@ -380,16 +409,28 @@ mlperf_backend_ptr_t SingleModelPipeline::backend_create( } bool use_gpu = strcmp(configs->delegate_selected, kDelegateGpu) == 0; +#if defined(__APPLE__) + use_gpu = use_gpu || strcmp(configs->delegate_selected, kDelegateMetal) == 0; +#endif + // Report an unrecognized selection before any platform downgrade below, so a + // valid choice that we deliberately fall back from is not logged as unknown. + if (!use_gpu && strcmp(configs->delegate_selected, kDelegateCpu) != 0) { + LOG(ERROR) << "Unknown delegate_selected: " << configs->delegate_selected + << "; using the CPU accelerator"; + } #if __ANDROID__ if (use_gpu && IsEmulator()) { LOG(INFO) << "Emulator detected, using the CPU accelerator"; use_gpu = false; } -#endif - if (!use_gpu && strcmp(configs->delegate_selected, kDelegateCpu) != 0) { - LOG(ERROR) << "Unknown delegate_selected: " << configs->delegate_selected - << "; using the CPU accelerator"; +#elif defined(__APPLE__) && TARGET_OS_SIMULATOR + // Only the ios_arm64 (device) slice of the Metal accelerator is downloaded by + // litert_backend.mk, so there is nothing for the simulator to dlopen. + if (use_gpu) { + LOG(INFO) << "Simulator detected, using the CPU accelerator"; + use_gpu = false; } +#endif if (use_gpu && backend_data->shards_num > 1) { // Two shards Run() concurrently in issue_query; on the shared LiteRT GPU diff --git a/mobile_back_litert/cpp/backend_litert/stable_diffusion_invoker.cc b/mobile_back_litert/cpp/backend_litert/stable_diffusion_invoker.cc index fb057b31f..09579678b 100644 --- a/mobile_back_litert/cpp/backend_litert/stable_diffusion_invoker.cc +++ b/mobile_back_litert/cpp/backend_litert/stable_diffusion_invoker.cc @@ -23,6 +23,7 @@ limitations under the License. #include "absl/log/log.h" #include "embedding_utils.h" #include "litert/cc/litert_tensor_buffer.h" +#include "litert_env.h" #include "sd_utils.h" namespace { @@ -99,7 +100,9 @@ bool ReadFloats(litert::TensorBuffer &buffer, std::vector *out, return true; } -// Run is synchronous; the buffers were created once at backend_create time. +// Run is synchronous. The buffers belong to the compiled model and are created +// with it -- at backend_create, or again on the platforms that release and +// rebuild a model between phases (see SDBackendData::set_phase). bool RunModel(SDModel &model, const char *what) { auto run = model.compiled->Run(kSignatureIndex, model.input_bufs, model.output_bufs); @@ -116,6 +119,16 @@ StableDiffusionInvoker::StableDiffusionInvoker(SDBackendData *backend_data) : backend_data_(backend_data) {} bool StableDiffusionInvoker::invoke(std::vector *image) { + // A no-op unless the platform keeps the three stages apart in memory (see + // SDBackendData). Rebuilding a model is cheap next to a query, and no extra + // compiles are paid for: the same three models are built either way. + if (backend_data_->set_phase && !backend_data_->set_phase(SDPhase::kEncode)) { + LOG(ERROR) << "Failed to prepare the text encoder"; + return false; + } + + LITERT_LOG_MEM("sd: query start"); + LOG(INFO) << "Prompt encoding started"; std::vector encoded_text; if (!encode_prompt(backend_data_->input_prompt_tokens, &encoded_text)) { @@ -127,6 +140,16 @@ bool StableDiffusionInvoker::invoke(std::vector *image) { return false; } + // Both prompts are encoded, and the contexts above are host vectors, so the + // encoder is finished with. The denoising loop is the memory peak, so it is + // not carried into it. + if (backend_data_->set_phase && + !backend_data_->set_phase(SDPhase::kDiffuse)) { + LOG(ERROR) << "Failed to prepare the diffusion model"; + return false; + } + LITERT_LOG_MEM("sd: switched to the diffusion phase"); + LOG(INFO) << "Diffusion process started"; std::vector latent; if (!diffusion_process(encoded_text, unconditional_encoded_text, @@ -135,8 +158,20 @@ bool StableDiffusionInvoker::invoke(std::vector *image) { return false; } + LITERT_LOG_MEM("sd: diffusion done"); + + // The decode needs neither the encoder nor the diffusion model, and on a + // constrained platform it cannot run alongside them. + if (backend_data_->set_phase && !backend_data_->set_phase(SDPhase::kDecode)) { + LOG(ERROR) << "Failed to prepare the decoder"; + return false; + } + LITERT_LOG_MEM("sd: switched to the decode phase"); + LOG(INFO) << "Image decoding started"; - return decode_image(latent, image); + const bool decoded = decode_image(latent, image); + LITERT_LOG_MEM("sd: decode done"); + return decoded; } bool StableDiffusionInvoker::encode_prompt(const std::vector &tokens, diff --git a/mobile_back_litert/cpp/backend_litert/stable_diffusion_pipeline.cc b/mobile_back_litert/cpp/backend_litert/stable_diffusion_pipeline.cc index 25709b36f..ae2dbb468 100644 --- a/mobile_back_litert/cpp/backend_litert/stable_diffusion_pipeline.cc +++ b/mobile_back_litert/cpp/backend_litert/stable_diffusion_pipeline.cc @@ -24,9 +24,15 @@ limitations under the License. #include #include +#if defined(__APPLE__) +#include +#endif + #include "absl/log/log.h" #include "embedding_utils.h" #include "litert/cc/litert_options.h" +#include "litert/cc/options/litert_gpu_options.h" +#include "litert_env.h" #include "stable_diffusion_invoker.h" namespace { @@ -55,6 +61,38 @@ struct TensorSpec { std::vector dims; }; +// The delegate names the settings may select. The Apple settings name the GPU +// choice after the accelerator that serves it, the same way the vision +// benchmarks and the TFLite backend's iOS settings do. +constexpr char kDelegateCpu[] = "CPU"; +constexpr char kDelegateGpu[] = "GPU"; +#if defined(__APPLE__) +constexpr char kDelegateMetal[] = "Metal"; +#endif + +// Free a compiled model and everything allocated from it. The buffers were +// created from the compiled model, so they have to go first -- the same +// ordering constraint the SDModel declaration order encodes. +// +// Called on every platform: by the GPU->CPU fallback when a compile fails +// partway through, and on Apple by the phase swap as well. +void ReleaseModel(SDModel *model) { + model->output_bufs.clear(); + model->input_bufs.clear(); + model->compiled.reset(); +} + +#if defined(__APPLE__) +// The filename of a model path. When a compile is killed by EXC_RESOURCE the +// process dies without unwinding, so the last line that reached the log is the +// only evidence of which of the three models was being built; a shared label +// would leave that unanswered. +std::string ModelLabel(const std::string &path) { + const size_t slash = path.rfind('/'); + return slash == std::string::npos ? path : path.substr(slash + 1); +} +#endif + bool backendExists = false; // The //flutter/cpp:utils config readers are deliberately not linked into @@ -148,12 +186,19 @@ bool CheckPackedSize(const litert::TensorBuffer &buffer, const TensorSpec &spec, return true; } -// Compiles one model on the CPU accelerator, binds every tensor by name, -// pins the batch dimension and creates the buffers that all invocations +// Compiles one model on the requested accelerator, binds every tensor by +// name, pins the batch dimension and creates the buffers that all invocations // reuse. On success `*input_indices` holds the signature index of each spec // in `input_specs`, in the same order. +// +// `use_gpu` asks for the GPU with CPU kept alongside it, so ops the delegate +// cannot take still run: the shipped exports are dynamic_int8 (text encoder, +// diffusion) and dynamic_fp16 (decoder), aimed at XNNPACK, and there is no +// fp32 export of them to switch to. Partial delegation is therefore the +// normal outcome here, not a failure. bool BuildModel(litert::Environment &env, const std::string &model_path, - int num_threads, const std::vector &input_specs, + int num_threads, bool use_gpu, + const std::vector &input_specs, const TensorSpec &output_spec, SDModel *model, std::vector *input_indices) { auto options = litert::Options::Create(); @@ -161,7 +206,59 @@ bool BuildModel(litert::Environment &env, const std::string &model_path, LOG(ERROR) << "Options::Create failed for " << model_path; return false; } - options->SetHardwareAccelerators(litert::HwAccelerators::kCpu); + if (use_gpu) { + options->SetHardwareAccelerators(litert::HwAccelerators::kGpu | + litert::HwAccelerators::kCpu); + auto gpu_options = options->GetGpuOptions(); + if (gpu_options) { + // Constant-tensor sharing is ON, and the models are built so it can be. + // + // The option decides whether the delegate materialises the weights or + // keeps them in their stored form and dequantises them in the shader. + // That is the difference between fitting on a phone and not: compiling + // the diffusion model on Metal costs 4409 MiB with sharing off and + // 1913 MiB with it on, against roughly 2885 MiB still available to this + // benchmark on an iPhone 16 Pro. iOS kills a process that crosses its + // per-process limit and the kill cannot be caught, so this is not a + // tuning knob here. + // + // Sharing used to be off because turning it on made the Metal backend + // emit a shader that does not compile: + // + // newLibraryWithSource: program_source:29:35: + // error: use of undeclared identifier 'scale' + // half4 w_scale_s0 = half4(float4(scale)); + // + // The backend carries two templates for dequantising int8 weights: one + // that reads a per-axis scale tensor, and one that takes a scalar. It + // picks the scalar form for per-tensor weights but never declares the + // arguments that form references. Single-op models place the fault + // precisely: per-tensor int8 FULLY_CONNECTED fails, while per-axis + // FULLY_CONNECTED, per-tensor CONV_2D, per-axis CONV_2D and fp16 CONV_2D + // all compile. tools/sd_gpu/convert.py therefore re-expresses those + // weights as per-axis, repeating the single scale they already carry, so + // the working template is chosen. The weights are untouched. + // + // The Metal accelerator is a prebuilt dylib, so the codegen itself + // cannot be patched from here; the model is the only side we control. + // + // AllowSrcQuantizedFcConvOps stays off. litert_gpu_options.h says + // sharing "must be true to use this", so it is now available -- but it + // is not needed (with statically shaped models the diffusion graph is + // fully delegated without it) and the header notes it quantizes the + // input tensors to 8 bit, which costs accuracy. + gpu_options->EnableConstantTensorSharing(true); + gpu_options->EnableAllowSrcQuantizedFcConvOps(false); + gpu_options->SetPrecision(litert::GpuOptions::Precision::kFp16); + gpu_options->SetBufferStorageType( + litert::GpuOptions::BufferStorageType::kBuffer); + } else { + LOG(WARNING) << "GetGpuOptions failed for " << model_path + << "; compiling with default GPU options"; + } + } else { + options->SetHardwareAccelerators(litert::HwAccelerators::kCpu); + } if (num_threads > 0) { auto cpu_options = options->GetCpuOptions(); if (cpu_options) { @@ -171,7 +268,13 @@ bool BuildModel(litert::Environment &env, const std::string &model_path, } } +#if defined(__APPLE__) + LITERT_LOG_MEM(("sd: compiling " + ModelLabel(model_path)).c_str()); +#endif auto compiled = litert::CompiledModel::Create(env, model_path, *options); +#if defined(__APPLE__) + LITERT_LOG_MEM(("sd: compiled " + ModelLabel(model_path)).c_str()); +#endif if (!compiled) { LOG(ERROR) << "CompiledModel::Create failed for " << model_path << ": " << compiled.Error().Message(); @@ -226,6 +329,9 @@ bool BuildModel(litert::Environment &env, const std::string &model_path, } model->input_bufs = std::move(*input_bufs); model->output_bufs = std::move(*output_bufs); +#if defined(__APPLE__) + LITERT_LOG_MEM(("sd: buffers ready " + ModelLabel(model_path)).c_str()); +#endif for (size_t i = 0; i < input_specs.size(); ++i) { if (!CheckPackedSize(model->input_bufs[(*input_indices)[i]], input_specs[i], @@ -291,18 +397,44 @@ mlperf_backend_ptr_t StableDiffusionPipeline::backend_create( const std::string decoder_path = dir + decoder_name; const std::string timestep_embeddings_path = dir + timestep_embeddings_name; - // The shipped exports are dynamic_int8 (text encoder, diffusion) and - // dynamic_fp16 (decoder), aimed at CPU/XNNPACK; the settings only offer a - // CPU delegate for this benchmark. - if (configs->delegate_selected != nullptr && - strcmp(configs->delegate_selected, "CPU") != 0) { - LOG(WARNING) << "Ignoring delegate_selected=" << configs->delegate_selected - << "; the Stable Diffusion pipeline runs on the CPU"; + bool use_gpu = false; + if (configs->delegate_selected != nullptr) { + use_gpu = strcmp(configs->delegate_selected, kDelegateGpu) == 0; +#if defined(__APPLE__) + use_gpu = + use_gpu || strcmp(configs->delegate_selected, kDelegateMetal) == 0; +#endif + // Report an unrecognized selection before the platform downgrade below, so + // a valid choice we deliberately fall back from is not logged as unknown. + if (!use_gpu && strcmp(configs->delegate_selected, kDelegateCpu) != 0) { + LOG(ERROR) << "Unknown delegate_selected: " << configs->delegate_selected + << "; using the CPU accelerator"; + } } +#if defined(__APPLE__) && TARGET_OS_SIMULATOR + // Only the ios_arm64 (device) slice of the Metal accelerator is downloaded + // by litert_backend.mk, so there is nothing for the simulator to dlopen. + if (use_gpu) { + LOG(INFO) << "Simulator detected, using the CPU accelerator"; + use_gpu = false; + } +#endif + +#if defined(__APPLE__) + // This is the most memory-hungry pipeline in the backend and, in a CI sweep, + // it starts after five other benchmarks have each built and torn down a + // backend. Hand back whatever the allocator is still holding from those + // before compiling about 1 GiB of models on top of it. + litert_apple::ReturnFreeMemoryToOS(); + LITERT_LOG_MEM("sd: before compiling models"); +#endif // One environment shared by all three compiled models, as the LiteRT - // header recommends. - auto env = litert::Environment::Create({}); + // header recommends. Built through the shared factory so the Apple runtime + // library directory is set the same way as in the other pipelines -- on iOS + // that directory is the only way the Metal accelerator is found at all, so + // the GPU choice below depends on it. + auto env = CreateLiteRtEnvironment(); if (!env) { LOG(ERROR) << "Environment::Create failed"; backend_delete(backend_data); @@ -331,32 +463,55 @@ mlperf_backend_ptr_t StableDiffusionPipeline::backend_create( }; const TensorSpec decoder_output = {"padded_conv2d_37", {1, 512, 512, 3}}; - std::vector indices; - if (!BuildModel(*backend_data->env, text_encoder_path, num_threads, - encoder_inputs, encoder_output, &backend_data->text_encoder, - &indices)) { - backend_delete(backend_data); - return nullptr; - } - backend_data->encoder_tokens_idx = indices[0]; - backend_data->encoder_positions_idx = indices[1]; + // All three models compile on the same accelerator or none of them do. A + // mixed pipeline would still produce images, but the benchmark reports a + // single accelerator name, and reporting one when two ran is worse than + // giving up the GPU. + auto build_all = [&](bool gpu) { + std::vector indices; + if (!BuildModel(*backend_data->env, text_encoder_path, num_threads, gpu, + encoder_inputs, encoder_output, &backend_data->text_encoder, + &indices)) { + return false; + } + backend_data->encoder_tokens_idx = indices[0]; + backend_data->encoder_positions_idx = indices[1]; - if (!BuildModel(*backend_data->env, diffusion_model_path, num_threads, - diffusion_inputs, diffusion_output, &backend_data->diffusion, - &indices)) { - backend_delete(backend_data); - return nullptr; - } - backend_data->diffusion_latent_idx = indices[0]; - backend_data->diffusion_context_idx = indices[1]; - backend_data->diffusion_timestep_idx = indices[2]; + if (!BuildModel(*backend_data->env, diffusion_model_path, num_threads, gpu, + diffusion_inputs, diffusion_output, + &backend_data->diffusion, &indices)) { + return false; + } + backend_data->diffusion_latent_idx = indices[0]; + backend_data->diffusion_context_idx = indices[1]; + backend_data->diffusion_timestep_idx = indices[2]; - if (!BuildModel(*backend_data->env, decoder_path, num_threads, decoder_inputs, - decoder_output, &backend_data->decoder, &indices)) { + if (!BuildModel(*backend_data->env, decoder_path, num_threads, gpu, + decoder_inputs, decoder_output, &backend_data->decoder, + &indices)) { + return false; + } + backend_data->decoder_latent_idx = indices[0]; + return true; + }; + + if (use_gpu && !build_all(true)) { + LOG(WARNING) << "GPU compilation failed; falling back to CPU"; + // A failed attempt can leave models compiled behind it, and on iOS those + // pages still count against the limit the CPU retry has to fit inside. + ReleaseModel(&backend_data->text_encoder); + ReleaseModel(&backend_data->diffusion); + ReleaseModel(&backend_data->decoder); +#if defined(__APPLE__) + litert_apple::ReturnFreeMemoryToOS(); +#endif + use_gpu = false; + } + if (!use_gpu && !build_all(false)) { backend_delete(backend_data); return nullptr; } - backend_data->decoder_latent_idx = indices[0]; + backend_data->accelerator = use_gpu ? "GPU" : "CPU"; if (!EmbeddingManager::getInstance().load_timestep_embeddings( timestep_embeddings_path)) { @@ -371,6 +526,90 @@ mlperf_backend_ptr_t StableDiffusionPipeline::backend_create( backend_data->unconditional_tokens[0] = kStartOfTextToken; backend_data->input_prompt_tokens.assign(kTokenCount, 0); + LITERT_LOG_MEM("sd: all three models compiled"); + +#if defined(__APPLE__) + // Keeping the stages apart only helps when releasing a model actually hands + // its memory back, and on the GPU it does not. ReleaseModel drops the LiteRT + // objects and ReturnFreeMemoryToOS empties libmalloc's free list, but the + // delegate's weights live in Metal buffers that libmalloc never owned, so + // nothing is returned. Measured on an iPhone 16 Pro, in MiB still available: + // + // all three models compiled 1517 + // query start, after releasing two of them 1586 (only 69 recovered) + // diffusion model rebuilt 72 -> killed + // + // Rebuilding therefore stacks a second copy on top of the first and the + // process crosses the limit. Compiled once and left alone the three cost + // 1368 MiB together and leave 1517 free, which is the whole working set with + // room to spare -- so on the GPU the phases are simply not used. + // + // The CPU path keeps them: there the weights are ordinary allocations, the + // release does return them, and the three models plus a phase's working set + // do not fit together. + if (!use_gpu) { + // See the comment on these members in the header. Each stage of a query + // needs exactly one of the three models, so only that one is kept compiled. + backend_data->set_phase = [backend_data, text_encoder_path, + diffusion_model_path, decoder_path, num_threads, + use_gpu, encoder_inputs, encoder_output, + diffusion_inputs, diffusion_output, + decoder_inputs, decoder_output](SDPhase phase) { + // Release first, then build, so two models are never resident + // at once. Destroying a model is not enough on its own: the pages + // stay on libmalloc's free list and keep counting against the limit + // until they are handed back, which is what makes the release + // visible to EXC_RESOURCE. + std::vector idx; + if (phase == SDPhase::kEncode) { + ReleaseModel(&backend_data->diffusion); + ReleaseModel(&backend_data->decoder); + litert_apple::ReturnFreeMemoryToOS(); + if (backend_data->text_encoder.compiled == nullptr) { + if (!BuildModel(*backend_data->env, text_encoder_path, num_threads, + use_gpu, encoder_inputs, encoder_output, + &backend_data->text_encoder, &idx)) { + return false; + } + backend_data->encoder_tokens_idx = idx[0]; + backend_data->encoder_positions_idx = idx[1]; + } + return true; + } + + if (phase == SDPhase::kDiffuse) { + ReleaseModel(&backend_data->text_encoder); + ReleaseModel(&backend_data->decoder); + litert_apple::ReturnFreeMemoryToOS(); + if (backend_data->diffusion.compiled == nullptr) { + if (!BuildModel(*backend_data->env, diffusion_model_path, num_threads, + use_gpu, diffusion_inputs, diffusion_output, + &backend_data->diffusion, &idx)) { + return false; + } + backend_data->diffusion_latent_idx = idx[0]; + backend_data->diffusion_context_idx = idx[1]; + backend_data->diffusion_timestep_idx = idx[2]; + } + return true; + } + + ReleaseModel(&backend_data->text_encoder); + ReleaseModel(&backend_data->diffusion); + litert_apple::ReturnFreeMemoryToOS(); + if (backend_data->decoder.compiled == nullptr) { + if (!BuildModel(*backend_data->env, decoder_path, num_threads, use_gpu, + decoder_inputs, decoder_output, &backend_data->decoder, + &idx)) { + return false; + } + backend_data->decoder_latent_idx = idx[0]; + } + return true; + }; + } +#endif + return backend_data; } @@ -402,6 +641,15 @@ void StableDiffusionPipeline::backend_delete(mlperf_backend_ptr_t backend_ptr) { // environment. Safe on a partially built backend: every member is RAII. delete static_cast(backend_ptr); backendExists = false; + LITERT_LOG_MEM("sd: backend deleted (before reclaim)"); +#if defined(__APPLE__) + // The next benchmark allocates into whatever this leaves behind, and this + // pipeline is the largest consumer in the backend. Destroying the models + // only returns the pages to the allocator's free list, where they still + // count against the limit, so hand them back to the OS here too. + litert_apple::ReturnFreeMemoryToOS(); + LITERT_LOG_MEM("sd: backend deleted (after reclaim)"); +#endif } // Run the inference for a sample. diff --git a/mobile_back_litert/cpp/backend_litert/stable_diffusion_pipeline.h b/mobile_back_litert/cpp/backend_litert/stable_diffusion_pipeline.h index 3342e9c26..49b892768 100644 --- a/mobile_back_litert/cpp/backend_litert/stable_diffusion_pipeline.h +++ b/mobile_back_litert/cpp/backend_litert/stable_diffusion_pipeline.h @@ -15,6 +15,7 @@ limitations under the License. #include #include +#include #include #include @@ -38,6 +39,16 @@ struct SDModel { size_t output_idx = 0; }; +// The three stages of a query, which on Apple are kept apart in memory. Each +// needs exactly one of the three models, and the values that pass between them +// -- the encoded prompts, then the latent -- are small host vectors, so no two +// models ever have to be resident at once. +enum class SDPhase { + kEncode, + kDiffuse, + kDecode, +}; + struct SDBackendData { const char *name = "LiteRT"; const char *vendor = "Google"; @@ -70,6 +81,33 @@ struct SDBackendData { // Host staging for the decoded image: backend_get_output hands out a // pointer into it, so it has to stay valid after the call returns. std::vector output; + + // Apple CPU only. Left null on the GPU, and everywhere off Apple, where + // every model stays resident. + // + // iOS kills a process that crosses a per-process limit (measured at 3376 MB + // on an 8 GB device) and the kill cannot be caught. Releasing a model and + // rebuilding it later is only worth doing when the release actually returns + // the memory, which is true of the CPU path and NOT of the GPU one: + // ReleaseModel drops the LiteRT objects and ReturnFreeMemoryToOS empties + // libmalloc's free list, but the delegate's weights sit in Metal buffers + // that libmalloc never owned. Measured on an iPhone 16 Pro, in MiB still + // available before the limit: + // + // before compiling models 2885 + // all three models compiled 1517 (the three cost 1368) + // query start, after releasing two of them 1586 (only 69 recovered) + // diffusion model rebuilt 72 -> killed + // + // So on the GPU the phases are not used at all: the three models are + // compiled once and left alone. They fit together with room for the working + // set, and not rebuilding them each query also takes a 20-step image from + // 45.5 s to 18.8 s on the macOS host. + // + // The CPU path keeps the phases. There the weights are ordinary allocations + // that the release does return, and the three models plus a phase's working + // set do not fit together. + std::function set_phase; }; // A pipeline for Stable Diffusion. diff --git a/mobile_back_litert/litert_backend.mk b/mobile_back_litert/litert_backend.mk index 5e734fa51..520862409 100644 --- a/mobile_back_litert/litert_backend.mk +++ b/mobile_back_litert/litert_backend.mk @@ -13,19 +13,45 @@ # limitations under the License. ########################################################################## -# LiteRT backend (Android only): the llm-* benchmarks run on the LiteRT -# compiled-model API, the vision/NLP benchmarks on the TFLite interpreter -# vendored inside LiteRT. +# LiteRT backend (Android and iOS): every benchmark runs on the LiteRT +# CompiledModel API -- the llm-* benchmarks on the LLM pipeline, the vision/NLP +# benchmarks on the single-model pipeline. +# The prebuilt accelerators are cached under a version-stamped directory. The +# iOS rule below reuses whatever is already there rather than re-downloading, +# so an unversioned path would silently keep an accelerator from an older +# LiteRT next to a newer runtime -- and the two do not load together: a 2.2.0 +# dylib will not load into a 2.1.5 runtime, nor the reverse. Bumping the +# version therefore has to change the cache path as well as the URL. +backend_litert_version=2.2.0 +backend_litert_bins_dir=output/litert-bins/${backend_litert_version} + +# The Metal accelerator is a prebuilt dylib that the Xcode project embeds +# unconditionally (it is dlopened at runtime, never linked), so it has to be +# downloaded for every iOS build, even with WITH_LITERT=0 where only the dummy +# backend is bundled. +backend_litert_ios_bin_filename=libLiteRtMetalAccelerator.dylib +backend_litert_ios_bins_url=https://storage.googleapis.com/litert/binaries/${backend_litert_version}/ios_arm64/${backend_litert_ios_bin_filename} +backend_litert_ios_file=${backend_litert_bins_dir}/${backend_litert_ios_bin_filename} +backend_litert_ios_lib_deps= mkdir -p ${backend_litert_bins_dir} && \ + { [ -s ${backend_litert_ios_file} ] || \ + curl -fSL --proto '=https' --retry 3 --retry-delay 5 \ + -o ${backend_litert_ios_file} ${backend_litert_ios_bins_url}; } + ifeq (${WITH_LITERT},1) $(info WITH_LITERT=1) - backend_litert_bins_dir=output/litert-bins backend_litert_bin_filename=libLiteRtClGlAccelerator.so - backend_litert_bins_url=https://storage.googleapis.com/litert/binaries/2.1.5/android_arm64/${backend_litert_bin_filename} + backend_litert_bins_url=https://storage.googleapis.com/litert/binaries/${backend_litert_version}/android_arm64/${backend_litert_bin_filename} backend_litert_lib_deps= mkdir -p ${backend_litert_bins_dir} && \ curl -fSL --proto '=https' -o ${backend_litert_bins_dir}/${backend_litert_bin_filename} ${backend_litert_bins_url} backend_litert_android_files=${BAZEL_LINKS_PREFIX}bin/mobile_back_litert/cpp/backend_litert/liblitertbackend.so \ ${backend_litert_bins_dir}/${backend_litert_bin_filename} backend_litert_android_target=//mobile_back_litert/cpp/backend_litert:liblitertbackend.so + backend_litert_ios_target=//mobile_back_litert/cpp/backend_litert/ios:liblitertbackend + backend_litert_ios_zip=${BAZEL_LINKS_PREFIX}bin/mobile_back_litert/cpp/backend_litert/ios/liblitertbackend.xcframework.zip backend_litert_filename=liblitertbackend +else + # xcode will give you an error if a backend is specified in xcode config but the file is missing + backend_litert_ios_target=//mobile_back_tflite/cpp/backend_dummy/ios:liblitertbackend + backend_litert_ios_zip=${BAZEL_LINKS_PREFIX}bin/mobile_back_tflite/cpp/backend_dummy/ios/liblitertbackend.xcframework.zip endif diff --git a/mobile_back_litert/tools/sd_gpu/README.md b/mobile_back_litert/tools/sd_gpu/README.md new file mode 100644 index 000000000..5753a2645 --- /dev/null +++ b/mobile_back_litert/tools/sd_gpu/README.md @@ -0,0 +1,155 @@ +# Stable Diffusion models for the LiteRT GPU delegate + +`convert.py` rewrites the three v5_0 Stable Diffusion exports so the LiteRT GPU +delegate accepts them. The published exports run on CPU only; with these +rewrites the diffusion model and the decoder are fully GPU-accelerated, which is +what makes `stable_diffusion` usable on Metal. + +The rewrites are shape and metadata changes plus dead-op removal. No weight is +touched, and the script refuses to write a model whose output is not +bit-identical to the original. + +## What blocks the GPU delegate + +Five separate things, all of which `convert.py` fixes -- four that stop the +delegate taking the graph at all, and one that stops the memory option the +delegate needs to fit on a phone: + +1. **Dynamic shapes.** Every input declares `-1` on the batch dimension and the + graphs recompute their own shapes at run time (`SHAPE` -> + `REDUCE_PROD`/`GATHER`/`PACK`/`CONCATENATION`/`BROADCAST_ARGS` -> + `RESHAPE`/`BROADCAST_TO`). A `RESHAPE` whose shape operand is computed has a + dynamic output and the delegate refuses the graph outright: + + ```text + Attempting to use a delegate that only supports static-sized tensors + with a graph that has dynamic-sized tensors + ``` + + The pipeline only ever runs batch 1, so those shapes are constants. + +2. **`BROADCAST_TO` is not implemented by the GPU delegate.** The group-norm + blocks use it to materialise both operands of a `MUL`/`SUB` to a common + shape, which TFLite's binary kernels already do implicitly. + +3. **Rank-5 tensors.** The group-norm blocks work in + `[1, H, W, groups, ch/group]` and the delegate refuses anything above rank 4. + Every rank-5 tensor in these models has a leading dimension of 1. + +4. **A stale `SUB` version.** TFLite raised `SUB` to version 3 because of those + rank-5 operands; the delegate supports up to version 2. After step 3 nothing + needs version 3, but the declared version stays and splits the graph into + partitions too small to delegate. This step is what takes the diffusion and + decoder graphs from ~7% delegated to fully accelerated. + +5. **Per-tensor int8 `FULLY_CONNECTED` weights.** These delegate fine, but they + make the Metal backend generate a shader that does not compile once constant + tensor sharing is enabled: + + ```text + newLibraryWithSource: program_source:29:35: + error: use of undeclared identifier 'scale' + half4 w_scale_s0 = half4(float4(scale)); + ``` + + The backend carries two templates for dequantising int8 weights -- one that + reads a per-axis scale tensor, one that takes a scalar -- and it picks the + scalar form for per-tensor weights without declaring the arguments that form + references. Single-op models place the fault exactly: per-tensor int8 + `FULLY_CONNECTED` fails, while per-axis `FULLY_CONNECTED`, per-tensor + `CONV_2D`, per-axis `CONV_2D` and fp16 `CONV_2D` all compile. + + Sharing is not optional here. It decides whether the delegate materialises + the weights or keeps them stored and dequantises them in the shader, and on + Metal the diffusion model costs 4409 MiB to compile without it against + 1913 MiB with it -- against roughly 2885 MiB available on an iPhone 16 Pro. + The accelerator is a prebuilt dylib, so the model is the only side we + control: re-expressing those weights as per-axis, repeating the one scale + they already carry, selects the template that compiles. 183 tensors in the + diffusion model and 72 in the text encoder are rewritten this way; the + decoder is fp16 and has none. + +## Results + +Measured on an M-series Mac through the LiteRT `CompiledModel` API, ms per +invocation: + +| model | original CPU | rewritten CPU | rewritten Metal | +| --- | --- | --- | --- | +| text encoder | 12.4 | 13.1 | 14.1 | +| diffusion model | 1494.0 | 977.0 | 331.8 | +| decoder | 7702.3 | 2000.5 | 463.1 | + +The diffusion model runs once per denoising step (20 by default) and the decoder +once per image, so for one image this is roughly 37.6 s against 7.1 s. + +Dropping `BROADCAST_TO` speeds up the CPU path too — it was materialising full +size tensors that the binary kernels now broadcast for free. + +That CPU gain does **not** show up end to end, and it does not carry back to +LiteRT 2.1.5 at all. Measured through the C++ backend on a macOS host, one +20-step image on CPU takes 43.84 s with the published models and 43.22 s with +the rewritten ones — a single query each, 1.4% apart, which is noise. The +rewrite is worth adopting for the delegate, not for the CPU path. + +These models also need **LiteRT 2.2.0 or newer**. On 2.1.5 the Metal backend +emits invalid shader source for the int8 diffusion weights (`use of undeclared +identifier 'q0'`) and the compile fails whatever the options say. + +The text encoder keeps two `GATHER`s on CPU: its token and position indices are +2-D and the delegate only accepts 1-D indices there. It costs ~13 ms against the +20 x 332 ms the diffusion model spends, so it is not worth reshaping around. + +## Usage + +```bash +uv run convert.py --in-dir --out-dir +``` + +The dependency pins are in the PEP 723 header of `convert.py`, so `uv` builds +the environment itself and nothing is installed into whatever Python the repo +otherwise uses. The `ai-edge-litert` pin is the point: it is the runtime whose +Metal accelerator these models are being made compatible with, and the +conversion should be reproduced against the same one. + +The input directory needs the three published exports: + +* `sd_text_encoder_dynamic_int8.tflite` +* `sd_diffusion_model_dynamic_int8.tflite` +* `sd_decoder_dynamic_fp16.tflite` + +The outputs are named `sd_*_litert.tflite`, matching the convention already used +by `llama_q8_ekv3072_litert.tflite`. + +Each conversion runs the original and the rewrite on the same random inputs and +prints `max_abs_diff`; anything other than `0.000e+00` aborts the write. + +Running the converter twice, in two separately built environments, produced +byte-identical files, which is why the backend settings can carry these md5s. +Note the limits of that: `ai-edge-litert` is pinned above but `numpy` and +`flatbuffers` are not, so this is evidence of determinism rather than a +guarantee of it. If a regenerated file does not match, download it instead of +assuming the checksum is stale. + +Note also what the equivalence check does **not** cover. It runs both models +through the CPU `Interpreter` on one seeded input set per model. That +establishes the rewrite computes the same function on that sample; it does not +exercise the fp16 Metal path these models are actually selected for, and it is +not a proof for all inputs. + +| file | md5 | +| --- | --- | +| `sd_text_encoder_litert.tflite` | `dd4041a27340e829dda3eb90928b0804` | +| `sd_diffusion_model_litert.tflite` | `6547cfadc83bd809969dcb90bf754efd` | +| `sd_decoder_litert.tflite` | `8aa94e17f9394958e0c71c653ab1140f` | + +## Hosting + +The three files are published under +`https://storage.googleapis.com/mlperf-mobile-public/litert/`, alongside +`llama_q8_ekv3072_litert.tflite`, and `litert_settings_apple.pbtxt` points its +`stable_diffusion` Metal choice at them. + +Anything republished here has to keep the checksums above in step, and +`listResources` walks *every* delegate choice: a URL that 404s blocks resource +preparation for stable diffusion rather than only degrading the Metal path. diff --git a/mobile_back_litert/tools/sd_gpu/convert.py b/mobile_back_litert/tools/sd_gpu/convert.py new file mode 100644 index 000000000..e1e8b4237 --- /dev/null +++ b/mobile_back_litert/tools/sd_gpu/convert.py @@ -0,0 +1,513 @@ +#!/usr/bin/env -S uv run --script +# /// script +# requires-python = ">=3.11" +# dependencies = [ +# "ai-edge-litert==2.2.0", +# "numpy==2.5.3", +# "flatbuffers==25.12.19", +# ] +# /// +# Copyright 2025 The MLPerf Authors. All Rights Reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# ============================================================================== +"""Rewrite the Stable Diffusion exports so the LiteRT GPU delegate accepts them. + +The v5_0 SD exports run only on CPU under LiteRT. Five separate things stop the +GPU delegate from taking them -- four that block delegation outright, and one +that blocks the memory option the delegate needs to fit on a phone. This script +fixes all five and checks that the result still computes exactly the same +function. + +1. Dynamic shapes. Every input declares -1 on the batch dimension and the + graphs recompute their own shapes at run time (SHAPE -> REDUCE_PROD / GATHER + / PACK / CONCATENATION / BROADCAST_ARGS -> RESHAPE / BROADCAST_TO). A RESHAPE + whose shape operand is computed has a dynamic output, and the delegate + refuses the whole graph: + + Attempting to use a delegate that only supports static-sized tensors + with a graph that has dynamic-sized tensors + + The pipeline only ever runs batch 1, so those shapes are constants. Pin the + inputs and constant-fold the shape cone. + +2. BROADCAST_TO is not implemented by the GPU delegate. The group-norm blocks + use it to materialise both operands of a MUL/SUB to a common shape, which + TFLite's binary kernels already do implicitly, so it can be deleted. + +3. Rank-5 tensors. The group-norm blocks work in [1, H, W, groups, ch/group] + and the delegate refuses anything above rank 4. Every rank-5 tensor here has + a leading dimension of 1, so dropping it is a shape-only change. + +4. A stale SUB version. TFLite raised SUB to version 3 because of those rank-5 + operands; the GPU delegate supports up to version 2. Once step 3 has run + nothing needs version 3, but the declared version stays and splits the graph + into partitions too small to delegate. This is the step that takes the + diffusion and decoder graphs from ~7% delegated to fully accelerated. + +5. Per-tensor int8 FULLY_CONNECTED weights. These delegate fine, but they make + the Metal backend generate a shader that references undeclared identifiers, + so compiling with constant tensor sharing enabled fails outright: + + newLibraryWithSource: error: use of undeclared identifier 'scale' + + Sharing is not optional here: without it the diffusion model costs 4409 MiB + to compile on Metal, against an iOS budget of about 2885 MiB. Re-expressing + those weights as per-axis, repeating the one scale they already carry, + selects a shader template that does compile and brings that down to 1913 MiB. + +Measured on an M-series Mac (LiteRT CompiledModel, ms per invocation): + + model original CPU rewritten CPU rewritten Metal + text encoder 12.4 13.1 14.1 + diffusion model 1494.0 977.0 331.8 + decoder 7702.3 2000.5 463.1 + +The diffusion and decoder graphs come out fully GPU-accelerated; the text +encoder keeps two GATHERs on CPU because its token/position indices are 2-D and +the delegate only takes 1-D indices there. All three outputs are bit-identical +to the originals. + +Usage: + uv run convert.py --in-dir --out-dir + +The dependency pins live in the PEP 723 header above, so uv builds the +environment itself and nothing has to be installed into the repo's Python. The +ai-edge-litert pin matters: it is the runtime whose Metal accelerator these +models are being made compatible with. + +The rewritten files are named *_litert.tflite, matching the convention already +used by llama_q8_ekv3072_litert.tflite. +""" +import argparse +import collections +import os +import sys + +import flatbuffers +import numpy as np +from ai_edge_litert import schema_py_generated as schema +from ai_edge_litert.interpreter import Interpreter + +import shape_eval + +# Ops whose own kernel broadcasts its operands, so an explicit BROADCAST_TO in +# front of them buys nothing. +BROADCASTING_CONSUMERS = { + "ADD", "SUB", "MUL", "DIV", "MAXIMUM", "MINIMUM", "POW", + "SQUARED_DIFFERENCE", "FLOOR_DIV", "FLOOR_MOD", "SELECT_V2", +} + +# Operator versions we may lower once the rank-5 tensors are gone, with the +# element types the older version covers. +VERSION_RULES = { + "SUB": (2, {schema.TensorType.FLOAT32, schema.TensorType.INT8, + schema.TensorType.UINT8, schema.TensorType.INT32}), +} + +SD_MODELS = [ + ("sd_text_encoder_dynamic_int8.tflite", "sd_text_encoder_litert.tflite"), + ("sd_diffusion_model_dynamic_int8.tflite", + "sd_diffusion_model_litert.tflite"), + ("sd_decoder_dynamic_fp16.tflite", "sd_decoder_litert.tflite"), +] + + +def ins_of(op): + return [] if op.inputs is None else [int(x) for x in op.inputs] + + +def outs_of(op): + return [] if op.outputs is None else [int(x) for x in op.outputs] + + +def serialize(model): + b = flatbuffers.Builder(1024) + b.Finish(model.Pack(b), file_identifier=b"TFL3") + return bytes(b.Output()) + + +def probe(model_bytes, batch=1): + """Resize inputs to `batch`, allocate, run once, and report every shape. + + The run matters. TFLite only fills in a dynamic tensor's dims while + executing, so without it anything downstream of a computed RESHAPE reports a + stale placeholder -- the text encoder's attention really produces + [12,77,64] but advertises [1,77,64]. experimental_preserve_all_tensors would + also give correct shapes but disables memory planning, which costs many + gigabytes on these graphs. + """ + it = Interpreter(model_content=model_bytes) + want = {} + for d in it.get_input_details(): + want[d["index"]] = [batch if v == -1 else int(v) + for v in d["shape_signature"]] + it.resize_tensor_input(d["index"], want[d["index"]], strict=False) + it.allocate_tensors() + for d in it.get_input_details(): + it.set_tensor(d["index"], + np.zeros(want[d["index"]], dtype=d["dtype"])) + it.invoke() + shapes = {d["index"]: [int(x) for x in d["shape"]] + for d in it.get_tensor_details()} + names = {d["index"]: d["name"] for d in it.get_tensor_details()} + del it + return shapes, names, want + + +def fold_shapes(data, batch=1, rounds=4): + """Pin the inputs and constant-fold the shape-computation cone.""" + folded = dead = 0 + for _ in range(rounds): + model = schema.ModelT.InitFromPackedBuf(data, 0) + sub = model.subgraphs[0] + shapes, names, want = probe(data, batch) + + # The interpreter appends temporaries past the flatbuffer's tensor + # count; everything below it must still line up by name or the rewrite + # would silently retarget the wrong tensors. + for j, t in enumerate(sub.tensors): + fb = "" if t.name is None else bytes(t.name).decode() + if j in names and names[j] != fb: + raise RuntimeError( + "interpreter/flatbuffer tensor mismatch at %d" % j) + + for idx, dims in want.items(): + t = sub.tensors[idx] + t.shape = list(dims) + t.shapeSignature = None + shapes[idx] = list(dims) + + values, dead_ops, unhandled = shape_eval.evaluate(model, sub, shapes) + if not dead_ops: + if folded == 0: + data = serialize(model) # still need the pinned inputs + break + + for j in sorted(values): + t = sub.tensors[j] + v = np.asarray(values[j]).astype(shape_eval.NP_TYPE[t.type]) + buf = schema.BufferT() + buf.data = np.frombuffer(v.tobytes(), dtype=np.uint8) + model.buffers.append(buf) + t.buffer = len(model.buffers) - 1 + t.shape = list(v.shape) + t.shapeSignature = None + + counts = collections.Counter( + shape_eval.opcode_name(model, sub.operators[i]) for i in dead_ops) + sub.operators = [op for i, op in enumerate(sub.operators) + if i not in dead_ops] + for t in sub.tensors: + t.shapeSignature = None + + folded += len(values) + dead += len(dead_ops) + print(" folded %d tensors, dropped %d ops %s%s" + % (len(values), len(dead_ops), dict(counts.most_common(8)), + (" UNHANDLED=%s" % unhandled) if unhandled else "")) + data = serialize(model) + return data, folded, dead + + +def drop_broadcast_to(model, sub, shapes): + """Delete BROADCAST_TO ops whose consumers broadcast on their own.""" + consumers = collections.defaultdict(list) + for oi, op in enumerate(sub.operators): + for i in ins_of(op): + if i >= 0: + consumers[i].append(oi) + graph_outputs = set(int(x) for x in (sub.outputs or [])) + + remap, dead = {}, set() + for oi, op in enumerate(sub.operators): + if shape_eval.opcode_name(model, op) != "BROADCAST_TO": + continue + src, out = ins_of(op)[0], outs_of(op)[0] + if out in graph_outputs: + continue + cons = [shape_eval.opcode_name(model, sub.operators[c]) + for c in consumers[out]] + if not cons: + dead.add(oi) + continue + # An identity broadcast is always safe to drop; a real one only if + # every consumer broadcasts its own operands. + if shapes.get(src) != shapes.get(out): + if not all(c in BROADCASTING_CONSUMERS for c in cons): + continue + remap[out] = src + dead.add(oi) + + def resolve(t): # a chain of broadcasts can remap onto another removed one + seen = set() + while t in remap and t not in seen: + seen.add(t) + t = remap[t] + return t + + for oi, op in enumerate(sub.operators): + if oi in dead or op.inputs is None: + continue + op.inputs = [resolve(int(x)) if int(x) >= 0 else int(x) + for x in op.inputs] + sub.operators = [op for i, op in enumerate(sub.operators) if i not in dead] + return len(dead) + + +def squeeze_rank5(model, sub, shapes): + """Drop the leading 1 from every rank>=5 tensor and fix the ops around it.""" + targets = {i for i, s in shapes.items() if len(s) >= 5} + if not targets: + return 0, {} + if any(shapes[i][0] != 1 for i in targets): + raise RuntimeError("a rank-5 tensor does not have a leading 1") + + notes = collections.Counter() + # TFLite shares identical constants between ops, so the same axes or shape + # buffer can be reached from several operators -- all 60 group norms in the + # decoder point at one axes tensor. Adjust each buffer once; decrementing a + # reduce axis twice silently produces wrong results. + touched = set() + for op in sub.operators: + name = shape_eval.opcode_name(model, op) + ins, outs = ins_of(op), outs_of(op) + if not [i for i in ins + outs if i in targets]: + continue + + if name == "RESHAPE" and outs[0] in targets and len(ins) > 1: + t = sub.tensors[ins[1]] + if t.buffer in touched: + continue + buf = model.buffers[t.buffer] + v = np.frombuffer(bytes(buf.data), dtype=np.int32).copy() + if len(v) >= 5 and v[0] == 1: + touched.add(t.buffer) + buf.data = np.frombuffer(v[1:].tobytes(), dtype=np.uint8) + t.shape = [len(v) - 1] + notes["reshape-target"] += 1 + elif name in ("MEAN", "SUM", "REDUCE_MAX", "REDUCE_MIN", "REDUCE_PROD"): + t = sub.tensors[ins[1]] + if t.buffer in touched: + continue + touched.add(t.buffer) + buf = model.buffers[t.buffer] + v = np.frombuffer(bytes(buf.data), dtype=np.int32).copy() + nv = np.array([a - 1 if a > 0 else a for a in v], dtype=np.int32) + buf.data = np.frombuffer(nv.tobytes(), dtype=np.uint8) + notes["reduce-axes"] += 1 + + # Constant data is unchanged: dropping a leading 1 moves no element. + for i in sorted(targets): + t = sub.tensors[i] + cur = [] if t.shape is None else [int(x) for x in t.shape] + t.shape = cur[1:] if len(cur) >= 5 and cur[0] == 1 else shapes[i][1:] + t.shapeSignature = None + q = t.quantization + if q is not None and getattr(q, "quantizedDimension", 0): + q.quantizedDimension = max(0, int(q.quantizedDimension) - 1) + notes["quant-dim"] += 1 + return len(targets), dict(notes) + + +def lower_op_versions(model, sub, shapes): + """Lower operator versions the graph no longer needs (see module docstring).""" + users = collections.defaultdict(list) + for op in sub.operators: + users[int(op.opcodeIndex)].append(op) + + lowered = {} + for ci, oc in enumerate(model.operatorCodes): + code = max(oc.builtinCode, oc.deprecatedBuiltinCode) + rule = VERSION_RULES.get(shape_eval.OPNAME.get(code)) + if rule is None or oc.version <= rule[0]: + continue + target, ok_types = rule + safe = all( + len(shapes.get(i, [])) <= 4 and sub.tensors[i].type in ok_types + for op in users[ci] for i in ins_of(op) + outs_of(op) if i >= 0) + if safe: + lowered[shape_eval.OPNAME.get(code)] = (oc.version, target) + oc.version = target + return lowered + + +def widen_fc_quantization(model, sub): + """Step 5: re-express per-tensor FULLY_CONNECTED weights as per-axis. + + The Metal backend carries two shader templates for dequantising int8 + weights: one that reads a per-axis scale tensor, and one that takes a + scalar. For per-tensor weights it emits the scalar form but never declares + the arguments that form references, so with constant tensor sharing on the + generated shader does not compile: + + newLibraryWithSource: error: use of undeclared identifier 'scale' + half4 w_scale_s0 = half4(float4(scale)); + error: use of undeclared identifier 'zero_point' + + Repeating the single scale across the output-channel axis selects the + per-axis template instead. Every entry is the value the tensor already + carried, so the dequantised weights are identical -- only quantization + metadata changes here, never a weight byte. + + This is what lets the pipeline turn sharing on, and sharing is what keeps + the diffusion model inside the iOS per-process limit: compiling it on Metal + costs 4409 MiB without sharing against 1913 MiB with it, and the budget at + that point in a run is about 2885 MiB. Measured with single-op models, the + trigger is per-tensor int8 FULLY_CONNECTED specifically: per-axis + FULLY_CONNECTED, per-tensor CONV_2D, per-axis CONV_2D and fp16 CONV_2D all + compile. + """ + names = {} + for ci, oc in enumerate(model.operatorCodes): + code = max(oc.builtinCode, oc.deprecatedBuiltinCode) + names[ci] = shape_eval.OPNAME.get(code) + + widened = 0 + for op in sub.operators: + if names.get(int(op.opcodeIndex)) != "FULLY_CONNECTED": + continue + ins = ins_of(op) + if len(ins) < 2 or ins[1] < 0: + continue + weights = sub.tensors[ins[1]] + if weights.type != schema.TensorType.INT8: + continue + q = weights.quantization + # Already per-axis, or not quantized at all. A tensor shared by several + # FULLY_CONNECTED ops is widened by the first one and skipped here. + if q is None or q.scale is None or len(q.scale) != 1: + continue + if weights.shape is None or len(weights.shape) != 2: + continue + channels = int(weights.shape[0]) + zero = 0 if q.zeroPoint is None or len(q.zeroPoint) == 0 else int( + q.zeroPoint[0]) + q.scale = np.full(channels, float(q.scale[0]), dtype=np.float32) + q.zeroPoint = np.full(channels, zero, dtype=np.int64) + q.quantizedDimension = 0 + widened += 1 + return widened + + +def optimize(data): + """Steps 2-4: the op-level rewrites, after the shapes are static.""" + shapes, _, _ = probe(data) + model = schema.ModelT.InitFromPackedBuf(data, 0) + sub = model.subgraphs[0] + + n_bcast = drop_broadcast_to(model, sub, shapes) + n_sq, notes = squeeze_rank5(model, sub, shapes) + squeezed = {i: (v[1:] if len(v) >= 5 and v[0] == 1 else v) + for i, v in shapes.items()} + lowered = lower_op_versions(model, sub, squeezed) + n_widened = widen_fc_quantization(model, sub) + print(" dropped %d BROADCAST_TO, squeezed %d rank-5 tensors %s, " + "lowered %s, widened %d per-tensor FC weights" + % (n_bcast, n_sq, notes, lowered or "nothing", n_widened)) + return serialize(model) + + +def run(path_or_bytes, feeds): + if isinstance(path_or_bytes, bytes): + it = Interpreter(model_content=path_or_bytes) + else: + it = Interpreter(model_path=path_or_bytes) + for d in it.get_input_details(): + want = [1 if v == -1 else int(v) for v in d["shape_signature"]] + it.resize_tensor_input(d["index"], want, strict=False) + it.allocate_tensors() + for d in it.get_input_details(): + it.set_tensor(d["index"], feeds[d["name"]].astype(d["dtype"])) + it.invoke() + out = [np.array(it.get_tensor(d["index"])) for d in it.get_output_details()] + del it + return out + + +def verify(original_path, new_bytes, seed=0): + """The rewrite must not change what the model computes.""" + rng = np.random.default_rng(seed) + it = Interpreter(model_path=original_path) + feeds = {} + for d in it.get_input_details(): + shape = [1 if v == -1 else int(v) for v in d["shape_signature"]] + if np.issubdtype(d["dtype"], np.integer): + feeds[d["name"]] = rng.integers(0, 77, size=shape).astype(d["dtype"]) + else: + feeds[d["name"]] = rng.standard_normal(shape).astype(d["dtype"]) + del it + + a, b = run(original_path, feeds), run(new_bytes, feeds) + if len(a) != len(b): + print(" output count changed: %d -> %d" % (len(a), len(b))) + return False + + ok, worst = True, 0.0 + for x, y in zip(a, b): + if x.shape != y.shape: + print(" output shape changed: %s -> %s" % (x.shape, y.shape)) + ok = False + continue + # Compare the values themselves rather than reducing to a max + # difference. max(0.0, nan) is 0.0 in Python, so a NaN difference + # would otherwise report a clean 0.000e+00 and pass. + if not np.array_equal(x, y, equal_nan=True): + ok = False + d = np.abs(x.astype(np.float64) - y.astype(np.float64)) + finite = d[np.isfinite(d)] + if finite.size: + worst = max(worst, float(finite.max())) + if finite.size != d.size: + print(" non-finite values present in %d of %d elements" + % (d.size - finite.size, d.size)) + print(" verified against the original: max_abs_diff=%.3e%s" + % (worst, "" if ok else " MISMATCH")) + return ok + + +def convert_one(src, dst): + print(" %s" % os.path.basename(src)) + data = open(src, "rb").read() + data, folded, dead = fold_shapes(data) + data = optimize(data) + if not verify(src, data): + print(" REFUSING to write: output changed") + return False + with open(dst, "wb") as f: + f.write(data) + print(" wrote %s (%.1f MB)" % (dst, os.path.getsize(dst) / 1048576)) + return True + + +def main(): + ap = argparse.ArgumentParser(description=__doc__.split("\n")[0]) + ap.add_argument("--in-dir", required=True, + help="directory holding the v5_0 SD exports") + ap.add_argument("--out-dir", required=True) + args = ap.parse_args() + + os.makedirs(args.out_dir, exist_ok=True) + ok = True + for src_name, dst_name in SD_MODELS: + src = os.path.join(args.in_dir, src_name) + if not os.path.exists(src): + print(" missing %s -- skipped" % src) + ok = False + continue + ok &= convert_one(src, os.path.join(args.out_dir, dst_name)) + return 0 if ok else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/mobile_back_litert/tools/sd_gpu/shape_eval.py b/mobile_back_litert/tools/sd_gpu/shape_eval.py new file mode 100644 index 000000000..a050e236c --- /dev/null +++ b/mobile_back_litert/tools/sd_gpu/shape_eval.py @@ -0,0 +1,220 @@ +# Copyright 2025 The MLPerf Authors. All Rights Reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# ============================================================================== +"""Evaluate the shape-computation cone of a TFLite graph with numpy. + +`experimental_preserve_all_tensors` is the obvious way to learn what the shape +tensors hold, but it disables memory planning, so on the SD decoder and +diffusion exports it needs many gigabytes. Everything in the cone is small +integer arithmetic over tensor shapes, so evaluate it directly instead: SHAPE is +seeded from the (now fixed) shapes the interpreter reports, and the handful of +ops downstream of it are plain numpy. + +Ops that are not implemented here are simply not folded -- the graph stays +correct, it just stays dynamic, and the caller can see what was left behind. +""" +import numpy as np +from ai_edge_litert import schema_py_generated as schema + +OPNAME = {v: k for k, v in vars(schema.BuiltinOperator).items() + if isinstance(v, int)} + +NP_TYPE = { + schema.TensorType.FLOAT32: np.float32, + schema.TensorType.FLOAT16: np.float16, + schema.TensorType.INT32: np.int32, + schema.TensorType.INT64: np.int64, + schema.TensorType.INT8: np.int8, + schema.TensorType.UINT8: np.uint8, + schema.TensorType.INT16: np.int16, + schema.TensorType.BOOL: np.bool_, +} + + +def opcode_name(model, op): + oc = model.operatorCodes[op.opcodeIndex] + return OPNAME.get(max(oc.builtinCode, oc.deprecatedBuiltinCode), "?") + + +def const_value(model, sub, i): + """Return the constant value of tensor i, or None if it is not constant.""" + t = sub.tensors[i] + if not t.buffer: + return None + buf = model.buffers[t.buffer] + if buf.data is None or len(buf.data) == 0: + return None + dt = NP_TYPE.get(t.type) + if dt is None: + return None + shape = [] if t.shape is None else [int(x) for x in t.shape] + raw = np.frombuffer(bytes(buf.data), dtype=dt) + n = int(np.prod(shape)) if shape else 1 + if raw.size < n: + return None + return raw[:n].reshape(shape) if shape else raw[:1].reshape(()) + + +def _opts(op): + o = op.builtinOptions + return o if o is not None else None + + +def evaluate(model, sub, shapes, max_elements=4096): + """Return (values, dead_ops, unhandled) for the shape cone. + + values: {tensor index: np.ndarray} for tensors proven constant + dead_ops: indices of ops whose every output is in values + unhandled: Counter-ish dict of op names that blocked a fold + """ + values = {} + unhandled = {} + dead = set() + + def get(i): + if i in values: + return values[i] + return const_value(model, sub, i) + + for oi, op in enumerate(sub.operators): + name = opcode_name(model, op) + ins = [x for x in (op.inputs if op.inputs is not None else []) if x >= 0] + outs = [x for x in (op.outputs if op.outputs is not None else []) if x >= 0] + if not outs: + continue + + if name == "SHAPE": + if ins[0] not in shapes: + continue + dt = NP_TYPE.get(sub.tensors[outs[0]].type) + if dt is None: + unhandled["SHAPE(dtype)"] = unhandled.get("SHAPE(dtype)", 0) + 1 + continue + # Must match the tensor's declared type: writing int32 bytes into a + # tensor the schema calls int64 yields half the required bytes and + # TFLite rejects the model with "invalidly specified in schema". + values[outs[0]] = np.asarray(shapes[ins[0]], dtype=dt) + dead.add(oi) + continue + + argv = [get(i) for i in ins] + if any(a is None for a in argv): + continue + # Only fold small integer results: this pass exists to freeze shape + # vectors, not to bake activations into the file. + if any(sub.tensors[o].type not in (schema.TensorType.INT32, + schema.TensorType.INT64) + for o in outs): + continue + if any(int(np.prod(shapes.get(o, [1]) or [1])) > max_elements for o in outs): + continue + + o = _opts(op) + try: + if name == "GATHER": + axis = getattr(o, "axis", 0) or 0 + res = np.take(argv[0], argv[1].astype(np.int64), axis=axis) + elif name == "REDUCE_PROD": + axes = tuple(np.atleast_1d(argv[1]).astype(int).tolist()) + res = np.prod(argv[0], axis=axes, + keepdims=bool(getattr(o, "keepDims", False))) + elif name == "PACK": + axis = getattr(o, "axis", 0) or 0 + res = np.stack(argv, axis=axis) + elif name == "CONCATENATION": + axis = getattr(o, "axis", 0) or 0 + res = np.concatenate([np.atleast_1d(a) for a in argv], axis=axis) + elif name == "RESHAPE": + if len(argv) > 1: + res = argv[0].reshape([int(x) for x in np.atleast_1d(argv[1])]) + else: + res = argv[0].reshape([int(x) for x in (o.newShape or [])]) + elif name == "BROADCAST_ARGS": + res = np.asarray(np.broadcast_shapes( + tuple(int(x) for x in np.atleast_1d(argv[0])), + tuple(int(x) for x in np.atleast_1d(argv[1])))) + elif name == "CAST": + res = argv[0] + elif name in ("ADD",): + res = argv[0] + argv[1] + elif name in ("SUB",): + res = argv[0] - argv[1] + elif name in ("MUL",): + res = argv[0] * argv[1] + elif name in ("DIV", "FLOOR_DIV"): + res = argv[0] // argv[1] + elif name == "MAXIMUM": + res = np.maximum(argv[0], argv[1]) + elif name == "MINIMUM": + res = np.minimum(argv[0], argv[1]) + elif name == "EXPAND_DIMS": + res = np.expand_dims(argv[0], int(np.asarray(argv[1]).reshape(-1)[0])) + elif name == "SQUEEZE": + res = np.squeeze(argv[0]) + elif name == "RANGE": + res = np.arange(argv[0], argv[1], argv[2]) + elif name == "FILL": + res = np.full([int(x) for x in np.atleast_1d(argv[0])], argv[1]) + elif name == "TILE": + res = np.tile(argv[0], [int(x) for x in np.atleast_1d(argv[1])]) + elif name == "SLICE": + begin = [int(x) for x in np.atleast_1d(argv[1])] + size = [int(x) for x in np.atleast_1d(argv[2])] + sl = tuple(slice(b, None if s < 0 else b + s) + for b, s in zip(begin, size)) + res = argv[0][sl] + elif name == "STRIDED_SLICE": + begin = [int(x) for x in np.atleast_1d(argv[1])] + end = [int(x) for x in np.atleast_1d(argv[2])] + stride = [int(x) for x in np.atleast_1d(argv[3])] + bm = getattr(o, "beginMask", 0) or 0 + em = getattr(o, "endMask", 0) or 0 + sm = getattr(o, "shrinkAxisMask", 0) or 0 + if getattr(o, "ellipsisMask", 0) or getattr(o, "newAxisMask", 0): + raise NotImplementedError("strided_slice masks") + sl = [] + for d in range(len(begin)): + b = None if bm & (1 << d) else begin[d] + e = None if em & (1 << d) else end[d] + sl.append(slice(b, e, stride[d])) + res = argv[0][tuple(sl)] + for d in range(len(begin) - 1, -1, -1): + if sm & (1 << d): + res = np.take(res, 0, axis=d) + else: + unhandled[name] = unhandled.get(name, 0) + 1 + continue + except Exception as e: # noqa: BLE001 + unhandled["%s(%s)" % (name, type(e).__name__)] = \ + unhandled.get("%s(%s)" % (name, type(e).__name__), 0) + 1 + continue + + res = np.asarray(res) + if res.size > max_elements: + # Guard on the computed size too: an output whose shape was not in + # `shapes` skipped the pre-check above. + unhandled["%s(too-large)" % name] = \ + unhandled.get("%s(too-large)" % name, 0) + 1 + continue + want = shapes.get(outs[0]) + if want is not None and list(res.shape) != list(want): + # Our evaluation disagrees with the interpreter's own shape for this + # tensor -- do not trust it. + unhandled["%s(shape-mismatch)" % name] = \ + unhandled.get("%s(shape-mismatch)" % name, 0) + 1 + continue + values[outs[0]] = res.astype(NP_TYPE[sub.tensors[outs[0]].type]) + dead.add(oi) + + return values, dead, unhandled diff --git a/mobile_back_tflite/cpp/backend_dummy/ios/BUILD b/mobile_back_tflite/cpp/backend_dummy/ios/BUILD index a3229e059..860368222 100644 --- a/mobile_back_tflite/cpp/backend_dummy/ios/BUILD +++ b/mobile_back_tflite/cpp/backend_dummy/ios/BUILD @@ -28,8 +28,29 @@ apple_xcframework( "device": ["arm64"], }, minimum_os_versions = { - "ios": "13.1", - "macos": "13.1", + "ios": "15.0", + "macos": "15.0", + }, + deps = [ + "//mobile_back_tflite/cpp/backend_dummy:dummy_backend", + ], +) + +# This target is required to have the same name +# as the true litert backend. +# Otherwise xcode will fail to build the app. +apple_xcframework( + name = "liblitertbackend", + bundle_id = "com.mlcommons.inference.backend-litert-dummy", + exported_symbols_lists = ["//flutter/cpp/c:exported_symbols.lds"], + infoplists = ["//flutter/cpp/flutter:BackendBridgeInfo.plist"], + ios = { + "simulator": ["x86_64"], + "device": ["arm64"], + }, + minimum_os_versions = { + "ios": "15.0", + "macos": "15.0", }, deps = [ "//mobile_back_tflite/cpp/backend_dummy:dummy_backend", @@ -49,8 +70,8 @@ apple_xcframework( "device": ["arm64"], }, minimum_os_versions = { - "ios": "13.1", - "macos": "13.1", + "ios": "15.0", + "macos": "15.0", }, deps = [ "//mobile_back_tflite/cpp/backend_dummy:dummy_backend", diff --git a/mobile_back_tflite/cpp/backend_tflite/ios/BUILD b/mobile_back_tflite/cpp/backend_tflite/ios/BUILD index 0e7eb1452..0d6e3c5f8 100644 --- a/mobile_back_tflite/cpp/backend_tflite/ios/BUILD +++ b/mobile_back_tflite/cpp/backend_tflite/ios/BUILD @@ -14,8 +14,8 @@ apple_xcframework( "device": ["arm64"], }, minimum_os_versions = { - "ios": "13.1", - "macos": "13.1", + "ios": "15.0", + "macos": "15.0", }, deps = [ "//mobile_back_tflite/cpp/backend_tflite:tflite_c", diff --git a/patches/custom_buffer_teardown.patch b/patches/custom_buffer_teardown.patch index df2e5edd2..cf55b2ec8 100644 --- a/patches/custom_buffer_teardown.patch +++ b/patches/custom_buffer_teardown.patch @@ -1,10 +1,17 @@ -diff --git a/litert/runtime/custom_buffer.cc b/litert/runtime/custom_buffer.cc -index ca581e1..7764ab9 100644 --- a/litert/runtime/custom_buffer.cc +++ b/litert/runtime/custom_buffer.cc -@@ -20,0 +21 @@ +@@ -17,6 +17,7 @@ + #include + + #include +#include "litert/c/internal/litert_logging.h" -@@ -75,5 +76,2 @@ LITERT_NO_CFI_CHECK CustomBuffer::~CustomBuffer() { + + #include "litert/c/internal/litert_tensor_buffer_registry.h" + #include "litert/c/litert_any.h" +@@ -72,12 +73,26 @@ + } // namespace + + LITERT_NO_CFI_CHECK CustomBuffer::~CustomBuffer() { - LITERT_ASSIGN_OR_ABORT(auto registry, GetTensorBufferRegistry(env_)); - LITERT_ASSIGN_OR_ABORT(auto custom_buffer_handlers, - registry->GetCustomHandlers(buffer_type_)); @@ -12,7 +19,7 @@ index ca581e1..7764ab9 100644 - custom_buffer_handlers.destroy_func(hw_memory_info_); + if (!hw_memory_info_) { + return; -@@ -80,0 +79,17 @@ LITERT_NO_CFI_CHECK CustomBuffer::~CustomBuffer() { + } + // The registry is owned by the LiteRtEnvironment; if the environment was + // destroyed first, `env_` dangles. Guard the lookups instead of aborting. + auto registry = GetTensorBufferRegistry(env_); @@ -30,3 +37,6 @@ index ca581e1..7764ab9 100644 + return; + } + custom_buffer_handlers->destroy_func(hw_memory_info_); + } + + LITERT_NO_CFI_CHECK Expected CustomBuffer::Lock( diff --git a/patches/litert_coreml_repeatedfield_resize.patch b/patches/litert_coreml_repeatedfield_resize.patch deleted file mode 100644 index 5cbc72a07..000000000 --- a/patches/litert_coreml_repeatedfield_resize.patch +++ /dev/null @@ -1,15 +0,0 @@ -RepeatedField exposes Resize(), not resize(). The float16 branch just below -is a std::string (float16value has type bytes), so its lowercase resize() is -correct and left alone. - ---- a/tflite/delegates/coreml/builders/convolution_op_builder.cc -+++ b/tflite/delegates/coreml/builders/convolution_op_builder.cc -@@ -181,7 +181,7 @@ - if (weights_->type == kTfLiteFloat32) { - auto* coreml_weights = - layer_->mutable_convolution()->mutable_weights()->mutable_floatvalue(); -- coreml_weights->resize(NumElements(weights_), 0); -+ coreml_weights->Resize(NumElements(weights_), 0); - - optimized_ops::Transpose(params, tfl_shape, weights_->data.f, - coreml_shape, diff --git a/patches/litert_logistic_fp16_msvc.patch b/patches/litert_logistic_fp16_msvc.patch new file mode 100644 index 000000000..b62485e85 --- /dev/null +++ b/patches/litert_logistic_fp16_msvc.patch @@ -0,0 +1,23 @@ +Compute the reference Logistic kernel in float so it builds under MSVC. + +The same change as patches/tf_logistic_fp16_msvc.patch, against LiteRT's own +vendored copy of the TFLite kernels. The two files are identical apart from +their include paths, and the Windows build compiles both +@org_tensorflow//tensorflow/lite/kernels:builtin_op_kernels and +@litert//tflite/kernels:builtin_op_kernels, so both need it. + +diff --git a/tflite/kernels/internal/reference/logistic.h b/tflite/kernels/internal/reference/logistic.h +--- a/tflite/kernels/internal/reference/logistic.h ++++ b/tflite/kernels/internal/reference/logistic.h +@@ -44,7 +44,10 @@ + // optimized kernels. (check the definition of scalar_logistic_op) + + for (int i = 0; i < flat_size; i++) { +- T val = input_data[i]; ++ // Compute in float. For T = Eigen::half, comparing against the float ++ // cutoffs below is ambiguous under MSVC (error C2666): it sees both ++ // Eigen's half comparison operators and the builtin float one. ++ const float val = static_cast(input_data[i]); + float result; + if (val > cutoff_upper) { + result = 1.0f; diff --git a/patches/rules_cc_optional_cc_toolchain.patch b/patches/rules_cc_optional_cc_toolchain.patch new file mode 100644 index 000000000..57d09438a --- /dev/null +++ b/patches/rules_cc_optional_cc_toolchain.patch @@ -0,0 +1,48 @@ +Make the two legacy select()-support rules tolerate an unresolvable C++ +toolchain, restoring the behaviour of the rules_cc we used before LiteRT 2.2.0. + +rules_cc reaches us through rules_python's py_repositories(), by way of XLA's +and rules_ml_toolchain's python_init_repositories(). Moving TensorFlow to the +commit LiteRT 2.2.0 pins dragged that chain forward to rules_cc 0.1.5, whose +use_cc_toolchain() defaults to mandatory = True (0.0.16 defaulted to False). +compiler_flag and cc_flags_supplier therefore now demand a *registered* +toolchain for @bazel_tools//tools/cpp:toolchain_type. The Android build +resolves C++ toolchains the legacy way (--crosstool_top, with +--incompatible_enable_cc_toolchain_resolution=false in .bazelrc) and registers +none for @platforms//os:android, so analysis dies before anything is compiled: + + ERROR: external/rules_cc/cc/private/toolchain/BUILD:107:14: While resolving + toolchains for target @@rules_cc//cc/private/toolchain:compiler: No matching + toolchains found for types @@bazel_tools//tools/cpp:toolchain_type. + +which then cascades into "errors encountered resolving select() keys for +@@com_google_absl//absl/base:base", because Abseil's config_settings read +@bazel_tools//tools/cpp:compiler -- an alias for that target. + +Nothing is lost by making it optional. Both rules obtain their toolchain +through find_cpp_toolchain(), which checks the incompatible flag first and, +when resolution is disabled, reads the legacy _cc_toolchain attribute without +ever consulting ctx.toolchains. Where resolution *is* enabled -- iOS, macOS, +Windows, Linux -- a toolchain is found as before and mandatory changes nothing. + +diff --git a/cc/toolchains/compiler_flag.bzl b/cc/toolchains/compiler_flag.bzl +--- a/cc/toolchains/compiler_flag.bzl ++++ b/cc/toolchains/compiler_flag.bzl +@@ -23,5 +23,5 @@ + compiler_flag = rule( + implementation = _compiler_flag_impl, + attrs = CC_TOOLCHAIN_ATTRS, +- toolchains = use_cc_toolchain(), ++ toolchains = use_cc_toolchain(mandatory = False), + ) +diff --git a/cc/toolchains/cc_flags_supplier.bzl b/cc/toolchains/cc_flags_supplier.bzl +--- a/cc/toolchains/cc_flags_supplier.bzl ++++ b/cc/toolchains/cc_flags_supplier.bzl +@@ -28,6 +28,6 @@ + cc_flags_supplier = rule( + implementation = _cc_flags_supplier_impl, + attrs = CC_TOOLCHAIN_ATTRS, +- toolchains = use_cc_toolchain(), ++ toolchains = use_cc_toolchain(mandatory = False), + fragments = ["cpp"], + ) diff --git a/patches/tf_coreml_repeatedfield_resize.patch b/patches/tf_coreml_repeatedfield_resize.patch index 93d473fb1..b1d61c204 100644 --- a/patches/tf_coreml_repeatedfield_resize.patch +++ b/patches/tf_coreml_repeatedfield_resize.patch @@ -1,9 +1,10 @@ The protobuf bundled with this TensorFlow revision renames RepeatedField::resize() to Resize(), but the CoreML delegate still calls the -lowercase spelling for float32 weights. Same fix as -litert_coreml_repeatedfield_resize.patch, applied to TensorFlow's copy of -the delegate. The float16 branch just below is a std::string (float16value -has type bytes), so its lowercase resize() is correct and left alone. +lowercase spelling for float32 weights. TensorFlow keeps its own copy of the +delegate, which is why this patch is still needed; the equivalent LiteRT-side +patch was dropped because LiteRT 2.2.0 already calls Resize(). The float16 +branch just below is a std::string (float16value has type bytes), so its +lowercase resize() is correct and left alone. --- a/tensorflow/lite/delegates/coreml/builders/convolution_op_builder.cc +++ b/tensorflow/lite/delegates/coreml/builders/convolution_op_builder.cc diff --git a/patches/tf_logistic_fp16_msvc.patch b/patches/tf_logistic_fp16_msvc.patch new file mode 100644 index 000000000..36dfda5ae --- /dev/null +++ b/patches/tf_logistic_fp16_msvc.patch @@ -0,0 +1,34 @@ +Compute the reference Logistic kernel in float so it builds under MSVC. + +reference_ops::Logistic is instantiated for Eigen::half (activations.cc, +SigmoidEval on a kTfLiteFloat16 input) and compares that half against two float +cutoffs. The Eigen LiteRT 2.2.0's TensorFlow pins (ea13a98d, up from dcbaf2d6) +gives MSVC enough candidate operators to call the comparison ambiguous, and the +Windows build fails: + + external/org_tensorflow/tensorflow/lite/kernels/internal/reference/logistic.h(49): + error C2666: 'Eigen::ArrayBase::operator >': overloaded functions + have similar conversions + +Converting the input to float first removes the ambiguity and leaves behaviour +unchanged. For T = float this is the identity. For T = Eigen::half the compare +now happens in float rather than half, which only matters for inputs between +half(16.619047) == 16.609375 and 16.619047 -- and there the exact result, +0.99999994, still rounds to 1.0 in half. Both branches also already evaluated +in float: std::exp has no half overload, so the argument was converted anyway. + +diff --git a/tensorflow/lite/kernels/internal/reference/logistic.h b/tensorflow/lite/kernels/internal/reference/logistic.h +--- a/tensorflow/lite/kernels/internal/reference/logistic.h ++++ b/tensorflow/lite/kernels/internal/reference/logistic.h +@@ -44,7 +44,10 @@ + // optimized kernels. (check the definition of scalar_logistic_op) + + for (int i = 0; i < flat_size; i++) { +- T val = input_data[i]; ++ // Compute in float. For T = Eigen::half, comparing against the float ++ // cutoffs below is ambiguous under MSVC (error C2666): it sees both ++ // Eigen's half comparison operators and the builtin float one. ++ const float val = static_cast(input_data[i]); + float result; + if (val > cutoff_upper) { + result = 1.0f; diff --git a/patches/tf_portable_no_onednn_env_vars.patch b/patches/tf_portable_no_onednn_env_vars.patch index 78fc52ef7..2845667f5 100644 --- a/patches/tf_portable_no_onednn_env_vars.patch +++ b/patches/tf_portable_no_onednn_env_vars.patch @@ -1,5 +1,5 @@ -Two fixes to portable_tensorflow_lib_lite (the mobile build of TF core), -which has bit-rotted at this TensorFlow revision: +Fixes to portable_tensorflow_lib_lite (the mobile build of TF core), which +has bit-rotted at this TensorFlow revision: 1. It depends on //tensorflow/core/util:onednn_env_vars, a oneDNN/MKL helper whose deps unconditionally include the full //tensorflow/core:framework. @@ -16,15 +16,103 @@ which has bit-rotted at this TensorFlow revision: 3. Its quantization sources reach xla/tsl/concurrency/executor.h, which is not in its declared dependency closure; add the header-only target. +4. Two families of header are included but not declared, so the Android build + fails with "undeclared inclusion(s) in rule": + + xla/tsl/platform/file_system.h is in the mobile header set and includes + xla/tsl/platform/status_macros.h, which is not; that in turn reaches + absl/status/status_macros.h, status_builder.h and + absl/strings/internal/stringify_stream.h. Add the header to the filegroup + beside status.h and statusor.h, and depend on + @com_google_absl//absl/status:status_macros, whose deps cover the rest. + This appears only with the Abseil that LiteRT 2.2.0's TensorFlow pins, + which is where absl/status:status_macros first exists. + + tensorflow/core/kernels/pooling_ops_common.cc reaches + xla/tsl/framework/fixedpoint/*.h; depend on the header-only + @xla//xla/tsl/framework/fixedpoint that already exports them. + +5. Three mobile *hdrs* filegroups list proto rules -- ":example_protos_cc" in + tensorflow/core/example, and ":tensor_proto_cc", ":tensor_shape_proto_cc", + ":allocation_description_proto_cc", ":attr_value_proto_cc", + ":node_def_proto_cc", ":op_def_proto_cc" and ":types_proto_cc" in + tensorflow/core/framework. A rule in a filegroup expands to every one of its + outputs, so each generated .pb.cc rides in alongside its .pb.h, reaches + mobile_srcs, and is compiled into portable_tensorflow_lib_lite. The same + objects are also linked from the proto archives the target pulls in through + tf_portable_proto_lib(), so linking libbackendbridge.so fails on every + symbol in them: + + ld.lld: error: duplicate symbol: tensorflow::Example::clear_features() + >>> ... in archive .../tensorflow/core/example/libexample_protos.a + >>> ... _objs/portable_tensorflow_lib_lite/example.pb.o + + Drop them from the filegroups; the generated headers still arrive through + the proto libraries the target already depends on. The previous revision + listed only feature_util.{cc,h} in the example filegroup and had no such + entries, which is why this is new with the 2.2.0 pin. + +diff --git a/tensorflow/core/BUILD b/tensorflow/core/BUILD --- a/tensorflow/core/BUILD +++ b/tensorflow/core/BUILD -@@ -1036,7 +1036,8 @@ +@@ -1087,9 +1087,12 @@ "//tensorflow/core:mobile_additional_lib_deps", "//tensorflow/core/platform:resource", "//tensorflow/core/public:release_version", - "//tensorflow/core/util:onednn_env_vars", "//tensorflow/core/util:stats_calculator_portable", "@xla//xla/tsl/util:safe_reinterpret_cast", ++ "@com_google_absl//absl/status:status_macros", + "@com_google_protobuf//:protobuf", + "@xla//xla/tsl/concurrency:executor", ++ "@xla//xla/tsl/framework/fixedpoint", ] + tf_portable_proto_lib() + tf_portable_deps_no_runtime(), + alwayslink = 1, + ) +diff --git a/tensorflow/core/example/BUILD b/tensorflow/core/example/BUILD +--- a/tensorflow/core/example/BUILD ++++ b/tensorflow/core/example/BUILD +@@ -160,7 +160,6 @@ + name = "mobile_hdrs_no_runtime", + srcs = [ + "feature_util.h", +- ":example_protos_cc", + ], + ) + +diff --git a/tensorflow/core/framework/BUILD b/tensorflow/core/framework/BUILD +--- a/tensorflow/core/framework/BUILD ++++ b/tensorflow/core/framework/BUILD +@@ -366,8 +366,6 @@ + "variant_encode_decode.h", + "variant_op_registry.h", + "variant_tensor_data.h", +- ":tensor_proto_cc", +- ":tensor_shape_proto_cc", + "@xla//xla/tsl/framework:device_type.h", + ], + ) +@@ -478,12 +476,7 @@ + "tensor_util.h", + "thread_factory.h", + "versions.h", +- ":allocation_description_proto_cc", +- ":attr_value_proto_cc", + ":attr_value_proto_text_hdrs", +- ":node_def_proto_cc", +- ":op_def_proto_cc", +- ":types_proto_cc", + "//tensorflow/core/framework/registration:options.h", + "//tensorflow/core/framework/registration:registration.h", + ], +diff --git a/third_party/xla/xla/tsl/platform/BUILD b/third_party/xla/xla/tsl/platform/BUILD +--- a/third_party/xla/xla/tsl/platform/BUILD ++++ b/third_party/xla/xla/tsl/platform/BUILD +@@ -202,6 +202,7 @@ + "ram_file_system.h", + "resource.h", + "status.h", ++ "status_macros.h", + "statusor.h", + "threadpool.h", + "threadpool_interface.h",