From a6f16fff717fcfcaa5f7024d30e8b0cf3e2afd95 Mon Sep 17 00:00:00 2001 From: Scott Roy Date: Sun, 4 Oct 2026 19:50:32 -0700 Subject: [PATCH] [CoreAI] Integrate the delegate execute path Summary: Wire the Core AI bridge into an ExecuTorch `BackendInterface` so `CoreAIBackend` can load and run models. - `runtime/coreai_backend.mm`: the `BackendInterface`. `init` parses the manifest, checks the OS floor, selects the platform/architecture assets, acquires a prepared model through the load coordinator and binds the named function with ordered inputs/outputs. `execute` validates dtype (FP16/FP32), CPU storage, default dim order and contiguity, borrows input bytes without copying, then validates and resizes every output before copying any results. - Build: `backends/apple/coreai/CMakeLists.txt` builds `coreaidelegate` (static, or shared against `executorch_shared`) from the backend sources plus the `coreai_bridge_obj` objects, and links `coreai_swift`. The delegate and its test consumers link with the C++ linker, so ExecuTorch's C/C++ compile options never reach swiftc. - Tests: the fake bridge gains function binding, sessions and execution, plus the C entry points the backend calls, so the real backend runs in `coreai_host_test`. `coreai_runtime_smoke` checks final linkage, registration retention and availability; with `EXECUTORCH_BUILD_TESTS=ON` it is built and registered with CTest, otherwise it is excluded from the default build. - README: execution semantics, device-architecture selection, the default assets root, "Runtime options", and delegate linkage in "Building". `init` and `execute` block their calling thread on Swift concurrency work, so the README tells callers not to invoke them from Swift `async` code. Changes outside `backends/apple/coreai` are CMake only and, like the other backends, only take effect when `EXECUTORCH_BUILD_COREAI=ON`: - `tools/cmake/preset/default.cmake`: `EXECUTORCH_BUILD_COREAI` now requires the data loader extension. - `CMakeLists.txt`: add `coreaidelegate` to the backend list. Test Plan: ``` # Local: configure and build only cmake -S . -B build -G Ninja -DCMAKE_BUILD_TYPE=Release \ -DCMAKE_OSX_DEPLOYMENT_TARGET=27.0 -DEXECUTORCH_BUILD_COREAI=ON \ -DEXECUTORCH_BUILD_TESTS=ON -DEXECUTORCH_BUILD_EXTENSION_DATA_LOADER=ON cmake --build build --target backends/apple/coreai/all # CI (macOS 27 runner, .github/workflows/coreai.yml) ctest --test-dir build/backends/apple/coreai --output-on-failure --no-tests=error ``` CTest runs `coreai_host_test` (adding the delegate suite), `coreai_swift_bridge_test` and `coreai_runtime_smoke`. The smoke binary needs OS 27, so locally it is built but not run. --- CMakeLists.txt | 1 + backends/apple/coreai/CMakeLists.txt | 61 +- backends/apple/coreai/README.md | 107 +++ .../apple/coreai/runtime/coreai_backend.mm | 395 +++++++++++ .../test/coreai_acquisition_fixture.mm | 5 +- .../runtime/test/coreai_delegate_test.mm | 647 ++++++++++++++++++ .../coreai/runtime/test/coreai_fake_loader.h | 33 +- .../coreai/runtime/test/coreai_fake_loader.mm | 209 +++++- .../coreai/runtime/test/runtime_smoke.cpp | 27 + tools/cmake/preset/default.cmake | 4 + 10 files changed, 1480 insertions(+), 9 deletions(-) create mode 100644 backends/apple/coreai/runtime/coreai_backend.mm create mode 100644 backends/apple/coreai/runtime/test/coreai_delegate_test.mm create mode 100644 backends/apple/coreai/runtime/test/runtime_smoke.cpp diff --git a/CMakeLists.txt b/CMakeLists.txt index d64d972042f..71276d7bcb5 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -878,6 +878,7 @@ endif() if(EXECUTORCH_BUILD_COREAI) add_subdirectory(${CMAKE_CURRENT_SOURCE_DIR}/backends/apple/coreai) + list(APPEND _executorch_backends coreaidelegate) endif() if(EXECUTORCH_BUILD_MLX) diff --git a/backends/apple/coreai/CMakeLists.txt b/backends/apple/coreai/CMakeLists.txt index 0844ca245d3..e3100ff0147 100644 --- a/backends/apple/coreai/CMakeLists.txt +++ b/backends/apple/coreai/CMakeLists.txt @@ -22,8 +22,9 @@ endforeach() enable_language(OBJCXX) set(_coreai_runtime_sources - runtime/coreai_assets.mm runtime/coreai_storage.mm - runtime/coreai_bookmarks.mm runtime/coreai_load_coordinator.mm + runtime/coreai_backend.mm runtime/coreai_assets.mm + runtime/coreai_storage.mm runtime/coreai_bookmarks.mm + runtime/coreai_load_coordinator.mm ) # Keep ARC and C++ exception settings away from the private Swift module. @@ -57,14 +58,21 @@ if(EXECUTORCH_BUILD_TESTS) runtime/test/coreai_fake_loader.mm runtime/test/coreai_acquisition_test.mm runtime/test/coreai_acquisition_fixture.mm + runtime/test/coreai_delegate_test.mm runtime/ETCoreAITensor.mm ${_coreai_runtime_sources} ) coreai_configure_objc_target(coreai_host_test) target_compile_definitions(coreai_host_test PRIVATE COREAI_ASSETS_TESTING=1) find_library(COREAI_FOUNDATION_FRAMEWORK Foundation REQUIRED) + # The consolidated runtime target is created after this directory. + if(EXECUTORCH_BUILD_SHARED) + set(_coreai_host_runtime executorch_shared) + else() + set(_coreai_host_runtime executorch) + endif() target_link_libraries( - coreai_host_test PRIVATE executorch GTest::gtest + coreai_host_test PRIVATE ${_coreai_host_runtime} GTest::gtest ${COREAI_FOUNDATION_FRAMEWORK} ) add_test(NAME coreai_host_test COMMAND coreai_host_test) @@ -277,8 +285,53 @@ if(EXECUTORCH_BUILD_TESTS set_tests_properties(coreai_swift_bridge_test PROPERTIES TIMEOUT 120) endif() +if(EXECUTORCH_BUILD_SHARED) + set(_coreai_library_type SHARED) + set(_coreai_runtime executorch_shared) + set(_coreai_data_loader "") +else() + set(_coreai_library_type STATIC) + set(_coreai_runtime executorch) + set(_coreai_data_loader extension_data_loader) +endif() +add_library( + coreaidelegate ${_coreai_library_type} ${_coreai_runtime_sources} + $ +) +coreai_configure_objc_target(coreaidelegate) +# The delegate contains no Swift sources; its archive is assembled by C++. +set_target_properties(coreaidelegate PROPERTIES LINKER_LANGUAGE CXX) +coreai_require_os27(coreaidelegate) +target_include_directories(coreaidelegate PUBLIC ${_common_include_directories}) +add_dependencies(coreaidelegate coreai_swift) +target_link_libraries( + coreaidelegate PRIVATE coreai_swift ${_coreai_runtime} ${_coreai_data_loader} + executorch::coreai_dependencies +) +executorch_target_link_options_shared_lib(coreaidelegate) +if(EXECUTORCH_BUILD_SHARED) + set_target_properties( + coreaidelegate PROPERTIES OUTPUT_NAME executorch_backend_coreai + ) + executorch_target_soname_policy(coreaidelegate) + executorch_target_shipped_runtime_path(coreaidelegate) +endif() + +# A consumer is necessary to check Swift linkage and registration retention. +add_executable(coreai_runtime_smoke runtime/test/runtime_smoke.cpp) +if(EXECUTORCH_BUILD_TESTS) + add_test(NAME coreai_runtime_smoke COMMAND coreai_runtime_smoke) +else() + set_target_properties(coreai_runtime_smoke PROPERTIES EXCLUDE_FROM_ALL TRUE) +endif() +coreai_require_os27(coreai_runtime_smoke) +set_target_properties(coreai_runtime_smoke PROPERTIES LINKER_LANGUAGE CXX) +target_link_libraries( + coreai_runtime_smoke PRIVATE coreaidelegate ${_coreai_runtime} +) + install( - TARGETS coreai_swift + TARGETS coreaidelegate coreai_swift EXPORT ExecuTorchTargets LIBRARY DESTINATION ${CMAKE_INSTALL_LIBDIR} ARCHIVE DESTINATION ${CMAKE_INSTALL_LIBDIR} diff --git a/backends/apple/coreai/README.md b/backends/apple/coreai/README.md index f82a3860e1b..f6a33e79623 100644 --- a/backends/apple/coreai/README.md +++ b/backends/apple/coreai/README.md @@ -34,6 +34,34 @@ iteration order is not a binding contract. The `files` object maps relative filenames to byte sizes; `bundle_digests` maps each bundle basename to its export-time SHA-256 digest. +The experimental `CoreAIBackend` runtime supports stateless FP32/FP16 tensor +models in both formats. +ExecuTorch tensors must be contiguous. Output shapes +must fit the executor's declared resize and capacity constraints. Stateful +functions, image values, other dtypes, and interleaved layouts are not supported. +Calls on one session must be serialized by the caller. + +Backend `init` and `execute` block the calling thread until Core AI finishes, +and that work runs on Swift's shared concurrency pool. Do not load or execute +from Swift concurrency code (an `async` function or `Task`), because blocking +those threads can starve the pool; call from a dedicated thread or dispatch +queue instead. + +Supported inputs are borrowed directly from ExecuTorch storage through Core AI +raw views. The bridge copies shape metadata, not tensor bytes, and does not +allocate intermediate input `NSData`, `Data`, or `NDArray` buffers. Each call +wraps `const_data_ptr()` and the current shape's `nbytes()`, not the allocation's +maximum capacity. Outputs use the SDK's actual returned shape; the backend +resizes them within ExecuTorch's declared bounds before copying only the logical +result bytes. Unused capacity is neither wrapped as input nor written as output. +The backend waits for inference and all input access to finish before returning +from `execute`, including on errors. This uses ExecuTorch's normal input-lifetime +contract; callers must not concurrently mutate, release, or rebind the input +storage. Output data remains owned and is copied into validated ExecuTorch +outputs after inference, so output/input aliases are not written during the +borrow. This does not guarantee that Core AI itself avoids device transfers or +internal copies. + ### AOT architecture selection Configure AOT export with `AOTCompileConfig`, including the target `platform` and @@ -42,6 +70,14 @@ architectures=["h17p"])` requests that compiler architecture; omitting the list lets the compiler emit its supported architectures for the target platform. These are Core AI architecture names, not CPU names such as `arm64`. +At load time, the runtime queries `AIModel.deviceArchitectureName` and selects +the exact matching entry in the manifest's `archs` map. No runtime architecture +option is required, and there is no fallback to another architecture. Platform, +deployment floor, and architecture checks happen before asset reads or storage +creation. Missing-architecture errors identify the device architecture, list +available architectures, and request a matching export. Missing selected files +are reported separately from an absent architecture entry. + Both formats can require device specialization. Every load first attempts SDK bookmark restoration. A missing bookmark or SDK-confirmed cache miss falls back to source materialization and persistent specialization, then binds the requested @@ -49,6 +85,11 @@ function and ordered I/O on that same acquired model. ### Asset storage +Backend initialization reads the string runtime spec `coreai_assets_dir`. If +absent, it sets the directory to `executorch_coreai` beneath user-domain +`NSCachesDirectory`. The coordinator and storage helpers always receive that +explicit directory. It contains raw bookmarks, staged bundles and lock files. + The assets root must be an absolute path. The application chooses it, should reserve it for this backend, and is trusted not to rename, replace or modify its contents while the backend uses it; the backend does not defend against other @@ -83,6 +124,17 @@ Preparing the assets root sets `NSURLIsExcludedFromBackupKey` on it, which cover everything beneath it. Ancestors are not modified. This is backup exclusion, not a control for iCloud Drive synchronization. +Delegate construction and registration do not create storage directories or load +models. Per-model initialization performs SDK acquisition and function loading, +validating source assets only when recovery requires them. There is no synthetic +zero-input inference at initialization. SDK specialization and function resource +loading can still be substantial. + +The runtime always uses `AIModelCache.default` with persistent retention and +stores an opaque bookmark for direct restoration. `coreai_assets_dir` does not +choose the SDK artifact directory. The backend does not inspect SDK-private +files, alter their backup flags, or use an App Group cache. + Each partition derives a key from its selected export digest/bundle, platform, Core AI device architecture, default SDK cache namespace, versioned default options and persistent policy. Function bindings and delivery location are @@ -108,6 +160,36 @@ acquiring a lock does not flush its file or directory. Lock files are not removed by the backend, including after process exit. Removing the assets root externally requires all loads, sessions and maintenance to stop. +Automatic source removal is disabled. A durable bookmark and a live model pin +do not prove source independence. The backend does not remove sources on method +unload or delegate destruction; no runtime option enables automatic removal. +Cache storage and SDK persistent entries may be purged, so retain or redeliver +the original PTE and named data for reconstruction after a cache miss. +Source-free inference, fresh-process SDK restoration and persistent-policy +behavior require separate OS 27 hardware qualification. + +### Runtime options + +Pass options through the public `Module` API before loading. In an error-returning +loader using the `executorch::runtime` namespace: + +```cpp +BackendOptions<1> options; +ET_CHECK_OK_OR_RETURN_ERROR(options.set_option("coreai_assets_dir", assets_dir)); +LoadBackendOptionsMap map; +ET_CHECK_OK_OR_RETURN_ERROR(map.set_options("CoreAIBackend", options.view())); +ET_CHECK_OK_OR_RETURN_ERROR(module.load(map)); +``` + +No options are required when using the default assets root. `coreai_assets_dir` +selects an application-chosen cache directory reserved for this backend. The +supplied path is validated during preflight, but warm bookmark hits need no +source. + +Use separate option maps to choose different storage roots for different models. +The path is a root directory, not the bundle itself. ExecuTorch option strings +have a 255-byte limit. + ## SDK Bridge The private Swift module `CoreAIBridge` (`runtime/ETCoreAIModel.swift`) wraps @@ -140,6 +222,25 @@ Ninja builds use one architecture per build directory. Current runtime support covers arm64 macOS and iOS device builds. x86_64 builds are blocked by Swift `Float16` availability, and the tested iOS simulator SDKs do not contain Core AI. +Enable `EXECUTORCH_BUILD_EXTENSION_DATA_LOADER` for inference consumers. +Link the CMake target `coreaidelegate`, not just its archive filename. Its +transitive dependencies supply Swift and Foundation/CoreAI linkage, and its +link interface retains static backend registration. With +`EXECUTORCH_BUILD_SHARED=ON`, the delegate is a shared library linked to the +single consolidated `executorch_shared` runtime. Otherwise it is static. +Consumers of the shared delegate must use that same shared runtime. +Installed CMake targets resolve framework and Swift runtime dependencies using +the consumer's selected SDK and toolchain, not the producer's Xcode paths. +Select the Apple SDK/toolchain when configuring the consumer project. +SwiftPM and XCFramework distribution are not integrated yet. + +The `coreai_runtime_smoke` target checks final linkage of a C++ consumer, +including Swift dependencies and backend registration retention. With +`EXECUTORCH_BUILD_TESTS=ON` it is built and +registered with CTest; otherwise it is excluded from the default build. On OS 27, +running it checks registration and availability; it does not perform model +inference. Do not execute SDK27 binaries on an older host. + ## Host Tests `EXECUTORCH_BUILD_COREAI=ON` with `EXECUTORCH_BUILD_TESTS=ON` registers the @@ -159,3 +260,9 @@ cmake -S . -B -G Ninja \ cmake --build --target coreai_host_test ctest --test-dir -R '^coreai_host_test$' --output-on-failure ``` + +For a compile-only production smoke build, use a fresh build directory without +`EXECUTORCH_BUILD_TESTS`, set `CMAKE_OSX_ARCHITECTURES=arm64`, and build +`coreai_runtime_smoke`. For iOS also set `CMAKE_SYSTEM_NAME=iOS` and +`CMAKE_OSX_SYSROOT=iphoneos`. Do not run the iOS binary on the host or install +a simulator as part of this build. diff --git a/backends/apple/coreai/runtime/coreai_backend.mm b/backends/apple/coreai/runtime/coreai_backend.mm new file mode 100644 index 00000000000..01889c83e30 --- /dev/null +++ b/backends/apple/coreai/runtime/coreai_backend.mm @@ -0,0 +1,395 @@ +/* + * Copyright (c) Meta Platforms, Inc. and affiliates. + * All rights reserved. + * + * This source code is licensed under the BSD-style license found in the + * LICENSE file in the root directory of this source tree. + */ + +#import "ETCoreAIBridge.h" +#import "coreai_assets.h" +#import "coreai_load_coordinator.h" +#import "coreai_storage.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace executorch::backends::coreai { +namespace { +using namespace executorch::runtime; +using executorch::aten::ScalarType; +using executorch::aten::SizesType; +using executorch::aten::Tensor; + +struct Handle { + size_t input_count = 0; + size_t output_count = 0; + id session = nil; +}; + +Error bridge_error(NSError* error) { + ET_LOG(Error, "Core AI: %s", error.localizedDescription.UTF8String); + if ([error.domain isEqualToString:ETCoreAIErrorDomain]) { + switch (error.code) { + case ETCoreAIErrorInvalidModel: + return Error::InvalidProgram; + case ETCoreAIErrorUnsupported: + return Error::NotSupported; + case ETCoreAIErrorInvalidArgument: + return Error::InvalidArgument; + default: + break; + } + } + return Error::Internal; +} + +Result scalar_type(ScalarType type) { + switch (type) { + case ScalarType::Half: + return ETCoreAIScalarTypeFloat16; + case ScalarType::Float: + return ETCoreAIScalarTypeFloat32; + default: + ET_LOG(Error, "Core AI runtime supports FP16 and FP32 tensors only"); + return Error::NotSupported; + } +} + +Result tensor_bytes(const Tensor& tensor) { + ET_CHECK_OR_RETURN_ERROR( + tensor.device_type() == executorch::aten::DeviceType::CPU, + NotSupported, + "Core AI requires CPU-accessible tensor storage"); + ET_CHECK_OR_RETURN_ERROR( + tensor.dim() >= 0 && + static_cast(tensor.dim()) <= kTensorDimensionLimit, + InvalidArgument, + "Invalid Core AI tensor rank"); + ET_CHECK_OR_RETURN_ERROR( + tensor_is_default_dim_order(tensor), + NotSupported, + "Core AI requires default tensor dimension order"); + size_t elements = 1; + size_t expected_stride = 1; + for (size_t i = tensor.sizes().size(); i > 0; --i) { + const auto size = tensor.sizes()[i - 1]; + ET_CHECK_OR_RETURN_ERROR( + size >= 0, InvalidArgument, "Negative Core AI tensor dimension"); + ET_CHECK_OR_RETURN_ERROR( + tensor.strides()[i - 1] >= 0 && + static_cast(tensor.strides()[i - 1]) == expected_stride, + NotSupported, + "Core AI requires contiguous tensor storage"); + ET_CHECK_OR_RETURN_ERROR( + !c10::mul_overflows(elements, static_cast(size), &elements) && + !c10::mul_overflows( + expected_stride, + std::max(static_cast(size), size_t(1)), + &expected_stride), + InvalidArgument, + "Core AI tensor size overflow"); + } + size_t bytes = 0; + ET_CHECK_OR_RETURN_ERROR( + !c10::mul_overflows( + elements, static_cast(tensor.element_size()), &bytes) && + bytes <= static_cast(std::numeric_limits::max()) && + bytes == tensor.nbytes(), + InvalidArgument, + "Invalid Core AI tensor byte count"); + return bytes; +} + +Error validate_arguments( + Span args, + size_t input_count, + size_t output_count) { + ET_CHECK_OR_RETURN_ERROR( + args.size() >= input_count && + args.size() - input_count == output_count, + InvalidArgument, + "Core AI argument count mismatch"); + for (EValue* arg : args) { + ET_CHECK_OR_RETURN_ERROR( + arg && arg->isTensor(), + InvalidArgument, + "Core AI arguments must be tensors"); + auto type = scalar_type(arg->toTensor().scalar_type()); + if (!type.ok()) { + return type.error(); + } + auto bytes = tensor_bytes(arg->toTensor()); + if (!bytes.ok()) { + return bytes.error(); + } + } + return Error::Ok; +} + +Result*> borrow_inputs( + Span args, + size_t input_count) { + NSMutableArray* inputs = + [NSMutableArray arrayWithCapacity:input_count]; + for (size_t i = 0; i < input_count; ++i) { + const auto& tensor = args[i]->toTensor(); + ET_CHECK_OR_RETURN_ERROR( + tensor.nbytes() == 0 || tensor.const_data_ptr() != nullptr, + InvalidArgument, + "Core AI input has no storage"); + NSMutableArray* shape = + [NSMutableArray arrayWithCapacity:tensor.dim()]; + for (auto size : tensor.sizes()) { + [shape addObject:@(size)]; + } + [inputs addObject:[[ETCoreAIInputTensor alloc] + initWithBytes:tensor.const_data_ptr() + byteCount:tensor.nbytes() + shape:shape + scalarType:scalar_type(tensor.scalar_type()) + .get()]]; + } + return inputs; +} + +Error copy_outputs( + NSArray* outputs, + Span args, + size_t input_count, + size_t output_count) { + ET_CHECK_OR_RETURN_ERROR( + outputs != nil && outputs.count == output_count, + InvalidExternalData, + "Core AI output count mismatch"); + std::vector> shapes(output_count); + for (size_t i = 0; i < output_count; ++i) { + ETCoreAITensor* output = outputs[i]; + const auto& tensor = args[input_count + i]->toTensor(); + ET_CHECK_OR_RETURN_ERROR( + [output isKindOfClass:ETCoreAITensor.class] && + output.scalarType == scalar_type(tensor.scalar_type()).get() && + output.shape.count == static_cast(tensor.dim()), + InvalidExternalData, + "Core AI output type or rank mismatch"); + size_t bytes = tensor.element_size(); + for (id size in output.shape) { + ET_CHECK_OR_RETURN_ERROR( + [size isKindOfClass:NSNumber.class] && + [size longLongValue] >= 0 && + [size longLongValue] <= + std::numeric_limits::max() && + [size isEqualToNumber:@([size longLongValue])], + InvalidExternalData, + "Invalid Core AI output dimension"); + const auto dimension = static_cast([size longLongValue]); + shapes[i].push_back(dimension); + ET_CHECK_OR_RETURN_ERROR( + !c10::mul_overflows( + bytes, static_cast(dimension), &bytes), + InvalidExternalData, + "Core AI output size overflow"); + } + ET_CHECK_OR_RETURN_ERROR( + output.data != nil && bytes == output.data.length, + InvalidExternalData, + "Core AI output byte count mismatch"); + } + // Check all resized outputs before copying to avoid partial result writes. + for (size_t i = 0; i < output_count; ++i) { + auto& tensor = args[input_count + i]->toTensor(); + ET_CHECK_OK_OR_RETURN_ERROR(resize_tensor( + tensor, ArrayRef(shapes[i].data(), shapes[i].size()))); + auto bytes = tensor_bytes(tensor); + if (!bytes.ok()) { + return bytes.error(); + } + ET_CHECK_OR_RETURN_ERROR( + bytes.get() == outputs[i].data.length && + (bytes.get() == 0 || tensor.mutable_data_ptr() != nullptr), + InvalidArgument, + "Core AI output has insufficient storage"); + } + for (size_t i = 0; i < output_count; ++i) { + NSData* data = outputs[i].data; + if (data.length > 0) { + memcpy( + args[input_count + i]->toTensor().mutable_data_ptr(), + data.bytes, + data.length); + } + } + return Error::Ok; +} + +Result path_option(BackendInitContext& context, const char* key) { + auto option = context.get_runtime_spec(key); + if (!option.ok()) { + if (option.error() == Error::NotFound) { + return static_cast(nil); + } + return option.error(); + } + NSString* path = option.get() == nullptr + ? nil + : [NSString stringWithUTF8String:option.get()]; + ET_CHECK_OR_RETURN_ERROR( + path != nil && path.isAbsolutePath, + InvalidArgument, + "Core AI %s must be an absolute UTF-8 path", + key); + return path; +} + +class CoreAIBackend final : public BackendInterface { + public: + bool is_available() const override { return ETCoreAIIsAvailable(); } + + Result init( + BackendInitContext& context, + FreeableBuffer* processed, + ArrayRef) const override { + @autoreleasepool { + ET_CHECK_OR_RETURN_ERROR( + is_available(), NotSupported, "Core AI requires macOS 27 or iOS 27"); + ET_CHECK_OR_RETURN_ERROR( + processed && processed->data() && processed->size() > 0, + InvalidProgram, + "Missing Core AI manifest"); + NSData* bytes = + [NSData dataWithBytesNoCopy:const_cast(processed->data()) + length:processed->size() + freeWhenDone:NO]; + auto manifest = parse_manifest(bytes); + if (!manifest.ok()) { + return manifest.error(); + } + ET_CHECK_OR_RETURN_ERROR( + [NSProcessInfo.processInfo + isOperatingSystemAtLeastVersion:manifest->minimum_version], + DelegateInvalidCompatibility, + "Core AI model requires a newer operating system"); +#if TARGET_OS_OSX + NSString* platform = @"macOS"; +#elif TARGET_OS_IOS + NSString* platform = @"iOS"; +#else + NSString* platform = @"unsupported"; +#endif + NSString* architecture = ETCoreAIDeviceArchitectureName(); + auto selected = select_assets(manifest.get(), architecture, platform); + if (!selected.ok()) { + return selected.error(); + } + manifest.get() = std::move(selected.get()); + auto assets_option = path_option(context, "coreai_assets_dir"); + if (!assets_option.ok()) { + return assets_option.error(); + } + NSString* coreai_assets_dir = assets_option.get(); + if (coreai_assets_dir == nil) { + auto default_root = default_coreai_assets_root(); + if (!default_root.ok()) { + return default_root.error(); + } + coreai_assets_dir = default_root.get(); + } + id loader = ETCoreAICreateModelLoader(); + ET_CHECK_OR_RETURN_ERROR( + loader != nil, Internal, "Cannot create Core AI loader"); + __block id session = nil; + __block NSError* load_error = nil; + dispatch_semaphore_t ready = dispatch_semaphore_create(0); + auto prepared = acquire_bookmark_model( + manifest.get(), + context.get_named_data_map(), + coreai_assets_dir, + platform, + architecture, + loader); + if (!prepared.ok()) { + return prepared.error(); + } + [prepared.get() + loadFunctionNamed:manifest->function + inputNames:manifest->inputs + outputNames:manifest->outputs + completion:^(id result, NSError* error) { + session = result; + load_error = error; + dispatch_semaphore_signal(ready); + }]; + dispatch_semaphore_wait(ready, DISPATCH_TIME_FOREVER); + if (load_error != nil) { + session = nil; + return bridge_error(load_error); + } + ET_CHECK_OR_RETURN_ERROR( + session != nil, Internal, "Core AI returned no session"); + auto handle = std::unique_ptr(new (std::nothrow) Handle()); + ET_CHECK_OR_RETURN_ERROR( + handle != nullptr, + MemoryAllocationFailed, + "Cannot allocate Core AI handle"); + handle->input_count = manifest->inputs.count; + handle->output_count = manifest->outputs.count; + handle->session = session; + processed->Free(); + return static_cast(handle.release()); + } + } + + Error execute( + BackendExecutionContext&, + DelegateHandle* opaque, + Span args) const override { + @autoreleasepool { + ET_CHECK_OR_RETURN_ERROR( + opaque != nullptr, DelegateInvalidHandle, "Null Core AI handle"); + auto& handle = *static_cast(opaque); + const size_t input_count = handle.input_count; + const size_t output_count = handle.output_count; + ET_CHECK_OK_OR_RETURN_ERROR( + validate_arguments(args, input_count, output_count)); + auto inputs = borrow_inputs(args, input_count); + if (!inputs.ok()) { + return inputs.error(); + } + __block NSArray* outputs = nil; + __block NSError* execution_error = nil; + dispatch_semaphore_t ready = dispatch_semaphore_create(0); + [handle.session + executeInputs:inputs.get() + completion:^(NSArray* result, NSError* error) { + outputs = result; + execution_error = error; + dispatch_semaphore_signal(ready); + }]; + // ET may reuse input storage after execute returns, never before. + dispatch_semaphore_wait(ready, DISPATCH_TIME_FOREVER); + if (execution_error != nil) { + return bridge_error(execution_error); + } + return copy_outputs(outputs, args, input_count, output_count); + } + } + + void destroy(DelegateHandle* handle) const override { + @autoreleasepool { + delete static_cast(handle); + } + } +}; + +CoreAIBackend backend; +[[maybe_unused]] const auto registration = + register_backend({"CoreAIBackend", &backend}); +} // namespace +} // namespace executorch::backends::coreai diff --git a/backends/apple/coreai/runtime/test/coreai_acquisition_fixture.mm b/backends/apple/coreai/runtime/test/coreai_acquisition_fixture.mm index 154fd7e2ae4..fd4b2835e8b 100644 --- a/backends/apple/coreai/runtime/test/coreai_acquisition_fixture.mm +++ b/backends/apple/coreai/runtime/test/coreai_acquisition_fixture.mm @@ -207,8 +207,9 @@ ParallelAcquisitionObservation parallel_acquisitions(ScopedFakeBridgeState& brid loader = nil; } auto& state = bridge.state(); - const bool none_live = state.prepared_models.load() == 0 && state.loaders.load() == 0 && - state.missing_bundles.load() == 0; + const bool none_live = state.sessions.load() == 0 && state.prepared_models.load() == 0 && + state.loaders.load() == 0 && state.binding_mismatches.load() == 0 && + state.missing_bundles.load() == 0 && state.input_wait_timeouts.load() == 0; if (!send_child_byte(fd, static_cast(none_live))) _exit(12); // No GTest-bearing destructors run before InitGoogleTest, even on a failed child. _exit(close(fd) == 0 ? 0 : 13); diff --git a/backends/apple/coreai/runtime/test/coreai_delegate_test.mm b/backends/apple/coreai/runtime/test/coreai_delegate_test.mm new file mode 100644 index 00000000000..8ca06d48eeb --- /dev/null +++ b/backends/apple/coreai/runtime/test/coreai_delegate_test.mm @@ -0,0 +1,647 @@ +/* + * Copyright (c) Meta Platforms, Inc. and affiliates. + * All rights reserved. + * + * This source code is licensed under the BSD-style license found in the + * LICENSE file in the root directory of this source tree. + */ + +#include +#include +#include +#include +#include +#include +#include +#include + +#include "coreai_bookmark_fixture.h" +#import "coreai_fake_loader.h" +#include "coreai_source_fixture.h" + +namespace executorch::backends::coreai::testing { +namespace { +using namespace executorch::runtime; +using executorch::aten::ScalarType; +using executorch::aten::Tensor; +using executorch::aten::TensorImpl; + +class CoreAIDelegateTest : public ::testing::Test { + protected: + void SetUp() override { + backend = get_backend_class("CoreAIBackend"); + ASSERT_NE(backend, nullptr); + } + ScopedFakeBridgeState bridge; + const BackendInterface* backend = nullptr; +}; + +// Keep the caller-owned manifest alive across failed initialization attempts. +class DelegateFixture final { + public: + DelegateFixture(const BackendInterface* backend, ScopedFakeBridgeState& bridge, + NSDictionary* manifest = manifest_dict()) + : json(encode(manifest)), + processed( + json.bytes, json.length, + [](void* context, void*, size_t) { ++*static_cast(context); }, &freed), + backend_(backend), + bridge_(bridge) {} + + ~DelegateFixture() { + reset(); + EXPECT_TRUE(bridge_.wait_for_callbacks(dispatch_time(DISPATCH_TIME_NOW, 10 * NSEC_PER_SEC))); + processed.Free(); + if (json != nil) EXPECT_EQ(freed, 1); + EXPECT_EQ(data.requests.load(), data.releases.load()); + } + DelegateFixture(const DelegateFixture&) = delete; + DelegateFixture& operator=(const DelegateFixture&) = delete; + + Error init(NSString* root = nil) { + if (storage.url == nil || json == nil) return Error::Internal; + BackendOptions<1> options; + auto error = options.set_option("coreai_assets_dir", (root ?: storage.url.path).UTF8String); + if (error != Error::Ok) return error; + return init_with_options({options.view().data(), options.view().size()}); + } + + Error init_with_options(Span options, MemoryAllocator* allocator = nullptr) { + reset(); + BackendInitContext context(allocator, nullptr, "forward", &data, options); + auto result = backend_->init(context, &processed, {}); + if (!result.ok()) return result.error(); + handle = result.get(); + return Error::Ok; + } + + void reset() { + if (handle != nullptr) backend_->destroy(handle); + handle = nullptr; + } + + Error execute(Span args) { + BackendExecutionContext context; + return backend_->execute(context, handle, args); + } + + TestData data; + BookmarkDirectory storage; + int freed = 0; + NSData* json; + FreeableBuffer processed; + DelegateHandle* handle = nullptr; + + private: + const BackendInterface* backend_; + ScopedFakeBridgeState& bridge_; +}; + +struct FloatTensors { + float input[2] = {3, 7}; + float output[2] = {-1, -1}; + TensorImpl::SizesType input_size[1] = {2}, output_size[1] = {2}; + TensorImpl::DimOrderType order[1] = {0}; + TensorImpl::StridesType input_stride[1] = {1}, output_stride[1] = {1}; + TensorImpl input_impl{ScalarType::Float, 1, input_size, input, order, input_stride}; + TensorImpl output_impl{ScalarType::Float, + 1, + output_size, + output, + order, + output_stride, + TensorShapeDynamism::DYNAMIC_BOUND}; + EValue in{Tensor(&input_impl)}, out{Tensor(&output_impl)}; + EValue* args[2] = {&in, &out}; +}; + +template +struct RangeTensors { + static constexpr ScalarType dtype = + std::is_same_v ? ScalarType::Float : ScalarType::Half; + T input[12] = {}, output[12] = {}; + TensorImpl::SizesType input_shape[2] = {4, 2}, output_shape[2] = {4, 2}; + TensorImpl::DimOrderType order[2] = {0, 1}; + TensorImpl::StridesType input_strides[2] = {2, 1}, output_strides[2] = {2, 1}; + TensorImpl input_impl{ + dtype, 2, input_shape, input + 2, order, input_strides, TensorShapeDynamism::DYNAMIC_BOUND}; + TensorImpl output_impl{dtype, + 2, + output_shape, + output + 2, + order, + output_strides, + TensorShapeDynamism::DYNAMIC_BOUND}; + EValue in{Tensor(&input_impl)}, out{Tensor(&output_impl)}; + EValue* args[2] = {&in, &out}; + + RangeTensors() { + for (size_t i = 0; i < 12; ++i) input[i] = static_cast(i + 1); + fill_output(); + } + void fill_output() { std::fill_n(output, 12, static_cast(99)); } +}; + +// All preflight rows must fail before reading payloads or creating storage. +void expect_preflight_rejection(const BackendInterface* backend, ScopedFakeBridgeState& bridge, + NSDictionary* manifest, Error expected) { + DelegateFixture fixture(backend, bridge, manifest); + ASSERT_NE(fixture.storage.url, nil); + NSURL* absent = [fixture.storage.url URLByAppendingPathComponent:@"uncreated/models"]; + NSString* previous_bundle = bridge.state().last_bundle; + EXPECT_EQ(fixture.init(absent.path), expected); + EXPECT_EQ(fixture.data.attempts.load(), 0); + EXPECT_EQ(fixture.data.metadata_requests.load(), 0); + EXPECT_EQ(fixture.freed, 0); + EXPECT_EQ(bridge.state().last_bundle, previous_bundle); + EXPECT_FALSE([NSFileManager.defaultManager fileExistsAtPath:absent.path]); + NSError* error = nil; + NSArray* children = + [NSFileManager.defaultManager contentsOfDirectoryAtPath:fixture.storage.url.path + error:&error]; + EXPECT_NE(children, nil); + EXPECT_EQ(error, nil); + EXPECT_EQ(children.count, 0u); +} + +TEST_F(CoreAIDelegateTest, AvailabilityRejectsWithoutTakingManifestOwnership) { + @autoreleasepool { + DelegateFixture fixture(backend, bridge); + bridge.state().available = false; + EXPECT_FALSE(backend->is_available()); + EXPECT_EQ(fixture.init(), Error::NotSupported); + EXPECT_EQ(fixture.freed, 0); + EXPECT_EQ(fixture.data.attempts.load(), 0); + EXPECT_EQ(bridge.state().sessions.load(), 0); + bridge.state().available = true; + EXPECT_TRUE(backend->is_available()); + ASSERT_EQ(fixture.init(), Error::Ok); + EXPECT_EQ(fixture.freed, 1); + EXPECT_EQ(bridge.state().sessions.load(), 1); + } +} + +TEST_F(CoreAIDelegateTest, LoaderFactoryFailureLeavesManifestAndStorageUntouched) { + @autoreleasepool { + bridge.state().fail_loader_factory = true; + expect_preflight_rejection(backend, bridge, manifest_dict(), Error::Internal); + EXPECT_EQ(bridge.state().loaders.load(), 0); + EXPECT_EQ(bridge.state().sessions.load(), 0); + } +} + +TEST_F(CoreAIDelegateTest, PreflightRejectsBeforePayloadReadsOrStorageCreation) { + @autoreleasepool { + auto missing_function = aot_manifest_dict(); + [missing_function removeObjectForKey:@"function"]; + auto numeric_platform = aot_manifest_dict(); + numeric_platform[@"platform"] = @1; + auto escaping_arch = aot_manifest_dict(); + NSMutableDictionary* archs = [escaping_arch[@"archs"] mutableCopy]; + archs[@"arch_a"] = @"../model.arch_a.aimodelc"; + escaping_arch[@"archs"] = archs; + auto escaping_file = aot_manifest_dict(); + escaping_file[@"files"] = @{@"model.arch_a.aimodelc/../escape" : @1}; + auto ios = aot_manifest_dict(); + ios[@"platform"] = @"iOS"; + const struct { + const char* name; + NSMutableDictionary* manifest; + NSString* device_architecture; + Error expected; + } rows[] = { + {"missing function", missing_function, @"arch_b", Error::InvalidProgram}, + {"non-string platform", numeric_platform, @"arch_b", Error::InvalidProgram}, + {"escaping architecture path", escaping_arch, @"arch_b", Error::InvalidProgram}, + {"escaping file entry", escaping_file, @"arch_b", Error::InvalidProgram}, + {"incompatible platform", ios, @"arch_b", Error::DelegateInvalidCompatibility}, + {"missing device architecture", aot_manifest_dict(), nil, + Error::DelegateInvalidCompatibility}, + }; + for (const auto& row : rows) { + SCOPED_TRACE(row.name); + bridge.state().device_architecture = row.device_architecture; + ASSERT_NO_FATAL_FAILURE( + expect_preflight_rejection(backend, bridge, row.manifest, row.expected)); + } + bridge.state().device_architecture = @"arch_b"; + } +} + +TEST_F(CoreAIDelegateTest, PreflightRejectsFutureDeploymentForSourceAndAot) { + @autoreleasepool { + for (bool aot : {false, true}) { + SCOPED_TRACE(aot ? "AOT" : "source"); + auto dict = aot ? aot_manifest_dict() : manifest_dict(); + dict[@"min_deployment_version"] = @"999999.0"; + expect_preflight_rejection(backend, bridge, dict, Error::DelegateInvalidCompatibility); + } + } +} + +TEST_F(CoreAIDelegateTest, AotInitPassesOnlySelectedArchitectureBundleToSdk) { + @autoreleasepool { + NSString* arch = @"arch_a"; + DelegateFixture fixture(backend, bridge, aot_manifest_dict()); + aot_data(fixture.data, arch); + bridge.state().device_architecture = arch; + auto parsed = parse_manifest(fixture.json); + ASSERT_TRUE(parsed.ok()); + auto selected = select_assets(parsed.get(), arch, @"macOS"); + ASSERT_TRUE(selected.ok()); + ASSERT_EQ(fixture.init(), Error::Ok); + auto key = bookmark_key(selected.get(), @"macOS", arch); + ASSERT_TRUE(key.ok()); + NSString* expected = [fixture.storage.url.path + stringByAppendingPathComponent:[NSString stringWithFormat:@"staging/%@/%@", key.get(), + selected->path.lastPathComponent]]; + EXPECT_TRUE([bridge.state().last_bundle isEqualToString:expected]); + fixture.reset(); + EXPECT_EQ(bridge.state().sessions.load(), 0); + EXPECT_EQ(fixture.data.attempts.load(), 2); + EXPECT_EQ(fixture.data.releases.load(), fixture.data.requests.load()); + } +} + +TEST_F(CoreAIDelegateTest, InitBindingFailurePreservesManifestAndStagedSource) { + @autoreleasepool { + DelegateFixture fixture(backend, bridge); + bridge.state().fail_load = true; + EXPECT_EQ(fixture.init(), Error::InvalidProgram); + EXPECT_EQ(fixture.freed, 0); + EXPECT_EQ(bridge.state().sessions.load(), 0); + ASSERT_NE(bridge.state().last_bundle, nil); + EXPECT_TRUE([NSFileManager.defaultManager fileExistsAtPath:bridge.state().last_bundle]); + bridge.state().fail_load = false; + ASSERT_EQ(fixture.init(), Error::Ok); + EXPECT_EQ(fixture.freed, 1); + EXPECT_EQ(bridge.state().sessions.load(), 1); + EXPECT_TRUE([NSFileManager.defaultManager fileExistsAtPath:bridge.state().last_bundle]); + EXPECT_TRUE([bridge.state().last_bundle + hasPrefix:[fixture.storage.url.path stringByAppendingString:@"/"]]); + } + EXPECT_EQ(bridge.state().sessions.load(), 0); +} + +TEST_F(CoreAIDelegateTest, OrderedIoBindingUsesManifestFunctionAndNames) { + @autoreleasepool { + DelegateFixture fixture(backend, bridge); + ASSERT_EQ(fixture.init(), Error::Ok); + EXPECT_EQ(bridge.state().binding_mismatches.load(), 0); + FloatTensors tensors; + ASSERT_EQ(fixture.execute({tensors.args, 2}), Error::Ok); + EXPECT_EQ(tensors.output[0], 3); + EXPECT_EQ(tensors.output[1], 7); + EXPECT_EQ(bridge.state().last_input_bytes.load(), tensors.input); + EXPECT_EQ(bridge.state().executions.load(), 1); + } +} + +TEST_F(CoreAIDelegateTest, ArgumentValidationRejectsBeforeSdkExecution) { + @autoreleasepool { + DelegateFixture fixture(backend, bridge); + ASSERT_EQ(fixture.init(), Error::Ok); + FloatTensors tensors; + EXPECT_EQ(fixture.execute({tensors.args, 1}), Error::InvalidArgument); + EValue scalar(int64_t(3)); + EValue* args[] = {&scalar, &tensors.out}; + EXPECT_EQ(fixture.execute({args, 2}), Error::InvalidArgument); + EXPECT_EQ(bridge.state().executions.load(), 0); + } +} + +// No assertion may leave a caller thread blocked behind its input-read gate. +class BorrowedInputWorker final { + public: + BorrowedInputWorker(DelegateFixture& fixture, FloatTensors& tensors, FakeBridgeState& state) + : fixture_(fixture), tensors_(tensors), state_(state) { + state_.input_ready = dispatch_semaphore_create(0); + state_.allow_input_read = dispatch_semaphore_create(0); + } + ~BorrowedInputWorker() { + release_and_join(); + state_.input_ready = nil; + state_.allow_input_read = nil; + } + BorrowedInputWorker(const BorrowedInputWorker&) = delete; + BorrowedInputWorker& operator=(const BorrowedInputWorker&) = delete; + int start() { + const int error = pthread_create(&thread_, nullptr, run, this); + started_ = error == 0; + return error; + } + long wait_ready() { + return dispatch_semaphore_wait(state_.input_ready, + dispatch_time(DISPATCH_TIME_NOW, 5 * NSEC_PER_SEC)); + } + long wait_finished(int64_t timeout) { + return dispatch_semaphore_wait(finished_, dispatch_time(DISPATCH_TIME_NOW, timeout)); + } + void release() { dispatch_semaphore_signal(state_.allow_input_read); } + void release_and_join() { + release(); + if (started_) { + const long returned = + dispatch_semaphore_wait(returned_, dispatch_time(DISPATCH_TIME_NOW, 10 * NSEC_PER_SEC)); + if (returned != 0) { + ADD_FAILURE() << "Delegate worker did not return after releasing its input gate"; + // A stuck worker still borrows this stack; unwinding would be unsafe. + std::_Exit(EXIT_FAILURE); + } + EXPECT_EQ(pthread_join(thread_, nullptr), 0); + started_ = false; + } + } + Error status = Error::Internal; + + private: + static void* run(void* context) { + auto& worker = *static_cast(context); + @autoreleasepool { + worker.status = worker.fixture_.execute({worker.tensors_.args, 2}); + } + dispatch_semaphore_signal(worker.finished_); + dispatch_semaphore_signal(worker.returned_); + return nullptr; + } + DelegateFixture& fixture_; + FloatTensors& tensors_; + FakeBridgeState& state_; + dispatch_semaphore_t finished_ = dispatch_semaphore_create(0); + dispatch_semaphore_t returned_ = dispatch_semaphore_create(0); + pthread_t thread_{}; + bool started_ = false; +}; + +TEST_F(CoreAIDelegateTest, BorrowedInputStorageStaysLiveUntilSuccessOrError) { + @autoreleasepool { + DelegateFixture fixture(backend, bridge); + ASSERT_EQ(fixture.init(), Error::Ok); + FloatTensors tensors; + for (bool failure : {false, true}) { + SCOPED_TRACE(failure ? "runtime error" : "success"); + bridge.state().fail_execute = failure; + BorrowedInputWorker worker(fixture, tensors, bridge.state()); + const int started = worker.start(); + if (started != 0) { + worker.release_and_join(); + FAIL() << "pthread_create failed: " << started; + } + const long ready = worker.wait_ready(); + const void* observed = bridge.state().last_input_bytes.load(); + const long premature = worker.wait_finished(50 * NSEC_PER_MSEC); + worker.release(); + const long finished = worker.wait_finished(5 * NSEC_PER_SEC); + worker.release_and_join(); + ASSERT_TRUE(bridge.wait_for_callbacks(dispatch_time(DISPATCH_TIME_NOW, 10 * NSEC_PER_SEC))); + EXPECT_EQ(ready, 0); + EXPECT_EQ(observed, tensors.input); + EXPECT_NE(premature, 0); + EXPECT_EQ(finished, 0); + EXPECT_EQ(worker.status, failure ? Error::Internal : Error::Ok); + EXPECT_EQ(bridge.state().last_input_first_byte.load(), + reinterpret_cast(tensors.input)[0]); + } + } +} + +TEST_F(CoreAIDelegateTest, ExecuteMapsRuntimeErrors) { + @autoreleasepool { + DelegateFixture fixture(backend, bridge); + ASSERT_EQ(fixture.init(), Error::Ok); + FloatTensors tensors; + bridge.state().fail_execute = true; + EXPECT_EQ(fixture.execute({tensors.args, 2}), Error::Internal); + } +} + +TEST_F(CoreAIDelegateTest, MalformedOutputDoesNotWriteCallerStorage) { + @autoreleasepool { + DelegateFixture fixture(backend, bridge); + ASSERT_EQ(fixture.init(), Error::Ok); + FloatTensors tensors; + bridge.state().bad_output = true; + EXPECT_EQ(fixture.execute({tensors.args, 2}), Error::InvalidExternalData); + EXPECT_EQ(tensors.output[0], -1); + EXPECT_EQ(tensors.output[1], -1); + } +} + +TEST_F(CoreAIDelegateTest, NoncontiguousInputIsNotSupported) { + @autoreleasepool { + DelegateFixture fixture(backend, bridge); + ASSERT_EQ(fixture.init(), Error::Ok); + FloatTensors tensors; + tensors.input_stride[0] = 2; + EXPECT_EQ(fixture.execute({tensors.args, 2}), Error::NotSupported); + EXPECT_EQ(bridge.state().executions.load(), 0); + } +} + +TEST_F(CoreAIDelegateTest, NonemptyOutputRequiresStorage) { + @autoreleasepool { + DelegateFixture fixture(backend, bridge); + ASSERT_EQ(fixture.init(), Error::Ok); + FloatTensors tensors; + tensors.output_impl.set_data(nullptr); + EXPECT_EQ(fixture.execute({tensors.args, 2}), Error::InvalidArgument); + } +} + +TEST_F(CoreAIDelegateTest, OutputResizesDownAndBackWithinCapacity) { + @autoreleasepool { + DelegateFixture fixture(backend, bridge); + ASSERT_EQ(fixture.init(), Error::Ok); + FloatTensors tensors; + TensorImpl::SizesType one[] = {1}; + TensorImpl smaller(ScalarType::Float, 1, one, tensors.input, tensors.order, + tensors.input_stride); + EValue small{Tensor(&smaller)}; + EValue* args[] = {&small, &tensors.out}; + ASSERT_EQ(fixture.execute({args, 2}), Error::Ok); + EXPECT_EQ(tensors.out.toTensor().size(0), 1); + ASSERT_EQ(fixture.execute({tensors.args, 2}), Error::Ok); + EXPECT_EQ(tensors.out.toTensor().size(0), 2); + } +} + +TEST_F(CoreAIDelegateTest, OutputResizeFailurePreservesStorage) { + @autoreleasepool { + DelegateFixture fixture(backend, bridge); + ASSERT_EQ(fixture.init(), Error::Ok); + FloatTensors tensors; + TensorImpl::SizesType capacity[] = {1}; + TensorImpl output(ScalarType::Float, 1, capacity, tensors.output, tensors.order, + tensors.output_stride, TensorShapeDynamism::DYNAMIC_BOUND); + EValue too_small{Tensor(&output)}; + EValue* args[] = {&tensors.in, &too_small}; + tensors.output[0] = -2; + EXPECT_NE(fixture.execute({args, 2}), Error::Ok); + EXPECT_EQ(tensors.output[0], -2); + } +} + +TEST_F(CoreAIDelegateTest, HalfDtypePreservesBitsAndRejectsMixedOutputDtype) { + @autoreleasepool { + DelegateFixture fixture(backend, bridge); + ASSERT_EQ(fixture.init(), Error::Ok); + FloatTensors tensors; + uint16_t input[] = {0x3c00, 0x4000}, output[] = {0, 0}; + TensorImpl input_impl(ScalarType::Half, 1, tensors.input_size, input, tensors.order, + tensors.input_stride); + TensorImpl output_impl(ScalarType::Half, 1, tensors.output_size, output, tensors.order, + tensors.output_stride); + EValue in{Tensor(&input_impl)}, out{Tensor(&output_impl)}; + EValue* args[] = {&in, &out}; + ASSERT_EQ(fixture.execute({args, 2}), Error::Ok); + EXPECT_EQ(output[0], input[0]); + EXPECT_EQ(output[1], input[1]); + EXPECT_EQ(bridge.state().last_input_bytes.load(), input); + EValue* mixed[] = {&in, &tensors.out}; + EXPECT_EQ(fixture.execute({mixed, 2}), Error::InvalidExternalData); + } +} + +TEST_F(CoreAIDelegateTest, ScalarTensorsExecuteAndRejectOutputRankMismatch) { + @autoreleasepool { + DelegateFixture fixture(backend, bridge); + ASSERT_EQ(fixture.init(), Error::Ok); + FloatTensors tensors; + TensorImpl input(ScalarType::Float, 0, nullptr, tensors.input); + TensorImpl output(ScalarType::Float, 0, nullptr, tensors.output); + EValue in{Tensor(&input)}, out{Tensor(&output)}; + EValue* args[] = {&in, &out}; + ASSERT_EQ(fixture.execute({args, 2}), Error::Ok); + EXPECT_EQ(tensors.output[0], tensors.input[0]); + EXPECT_EQ(bridge.state().last_input_bytes.load(), tensors.input); + EValue* wrong_rank[] = {&tensors.in, &out}; + EXPECT_EQ(fixture.execute({wrong_rank, 2}), Error::InvalidExternalData); + } +} + +TEST_F(CoreAIDelegateTest, EmptyVectorAndMatrixBorrowNullStorage) { + @autoreleasepool { + DelegateFixture fixture(backend, bridge); + ASSERT_EQ(fixture.init(), Error::Ok); + for (int rank : {1, 2}) { + SCOPED_TRACE(rank); + TensorImpl::SizesType shape[] = {2, 0}; + TensorImpl::DimOrderType order[] = {0, 1}; + TensorImpl::StridesType strides[] = {1, 1}; + auto* sizes = rank == 1 ? shape + 1 : shape; + TensorImpl input(ScalarType::Float, rank, sizes, nullptr, order, strides); + TensorImpl output(ScalarType::Float, rank, sizes, nullptr, order, strides); + EValue in{Tensor(&input)}, out{Tensor(&output)}; + EValue* args[] = {&in, &out}; + ASSERT_EQ(fixture.execute({args, 2}), Error::Ok); + EXPECT_EQ(bridge.state().last_input_bytes.load(), nullptr); + } + } +} + +TEST_F(CoreAIDelegateTest, BorrowedStorageMayAliasOutputAfterExecution) { + @autoreleasepool { + DelegateFixture fixture(backend, bridge); + ASSERT_EQ(fixture.init(), Error::Ok); + FloatTensors tensors; + tensors.output_impl.set_data(tensors.input); + ASSERT_EQ(fixture.execute({tensors.args, 2}), Error::Ok); + EXPECT_EQ(tensors.input[0], 3); + EXPECT_EQ(tensors.input[1], 7); + } +} + +TEST_F(CoreAIDelegateTest, BorrowedStorageUsesCurrentInputPointer) { + @autoreleasepool { + DelegateFixture fixture(backend, bridge); + ASSERT_EQ(fixture.init(), Error::Ok); + FloatTensors tensors; + ASSERT_EQ(fixture.execute({tensors.args, 2}), Error::Ok); + float next_input[] = {11, 13}; + tensors.input_impl.set_data(next_input); + ASSERT_EQ(fixture.execute({tensors.args, 2}), Error::Ok); + EXPECT_EQ(bridge.state().last_input_bytes.load(), next_input); + EXPECT_EQ(tensors.output[0], 11); + EXPECT_EQ(tensors.output[1], 13); + } +} + +TEST_F(CoreAIDelegateTest, DynamicShapesBorrowCurrentSubrange) { + @autoreleasepool { + DelegateFixture fixture(backend, bridge); + ASSERT_EQ(fixture.init(), Error::Ok); + RangeTensors tensors; + for (TensorImpl::SizesType rows : {3, 0}) { + SCOPED_TRACE(rows); + TensorImpl::SizesType shape[] = {rows, 2}; + ASSERT_EQ(resize_tensor(tensors.in.toTensor(), {shape, 2}), Error::Ok); + tensors.fill_output(); + ASSERT_EQ(fixture.execute({tensors.args, 2}), Error::Ok); + const size_t elements = static_cast(rows) * 2; + EXPECT_EQ(bridge.state().last_input_bytes.load(), tensors.input + 2); + EXPECT_EQ(bridge.state().last_input_byte_count.load(), elements * sizeof(float)); + EXPECT_EQ(tensors.out.toTensor().size(0), rows); + EXPECT_EQ(tensors.out.toTensor().const_data_ptr(), tensors.output + 2); + for (size_t i = 0; i < 12; ++i) { + SCOPED_TRACE(i); + EXPECT_EQ(tensors.output[i], i >= 2 && i < 2 + elements ? tensors.input[i] : 99.0f); + } + } + } +} + +TEST_F(CoreAIDelegateTest, StorageOptionsRejectWrongTypeAndRelativePaths) { + @autoreleasepool { + DelegateFixture fixture(backend, bridge); + { + SCOPED_TRACE("integer assets directory"); + BackendOption wrong_type{"coreai_assets_dir", 1}; + EXPECT_EQ(fixture.init_with_options({&wrong_type, 1}), Error::InvalidArgument); + } + { + SCOPED_TRACE("relative assets directory"); + BackendOptions<1> relative; + ASSERT_EQ(relative.set_option("coreai_assets_dir", "relative"), Error::Ok); + EXPECT_EQ(fixture.init_with_options({relative.view().data(), relative.view().size()}), + Error::InvalidArgument); + } + EXPECT_EQ(fixture.freed, 0); + } +} + +TEST_F(CoreAIDelegateTest, LifetimeDestroyRetainsSourceAndReloadIsolatesRoots) { + @autoreleasepool { + DelegateFixture first(backend, bridge); + ASSERT_EQ(first.init(), Error::Ok); + EXPECT_EQ(first.freed, 1); + EXPECT_EQ(bridge.state().sessions.load(), 1); + NSString* first_bundle = bridge.state().last_bundle; + ASSERT_NE(first_bundle, nil); + first.reset(); + EXPECT_EQ(bridge.state().sessions.load(), 0); + EXPECT_TRUE([NSFileManager.defaultManager fileExistsAtPath:first_bundle]); + backend->destroy(nullptr); + + DelegateFixture reloaded(backend, bridge); + ASSERT_EQ(reloaded.init(first.storage.url.path), Error::Ok); + EXPECT_TRUE([bridge.state().last_bundle isEqualToString:first_bundle]); + DelegateFixture other(backend, bridge); + NSURL* nested = [other.storage.url URLByAppendingPathComponent:@"nested/models"]; + EXPECT_FALSE([NSFileManager.defaultManager fileExistsAtPath:nested.path]); + ASSERT_EQ(other.init(nested.path), Error::Ok); + EXPECT_EQ(bridge.state().sessions.load(), 2); + EXPECT_TRUE([bridge.state().last_bundle + hasPrefix:[other.storage.url.path stringByAppendingString:@"/"]]); + EXPECT_FALSE([bridge.state().last_bundle isEqualToString:first_bundle]); + EXPECT_TRUE([NSFileManager.defaultManager fileExistsAtPath:first_bundle]); + other.reset(); + reloaded.reset(); + EXPECT_EQ(bridge.state().sessions.load(), 0); + } + EXPECT_EQ(bridge.state().sessions.load(), 0); + EXPECT_EQ(bridge.state().prepared_models.load(), 0); + EXPECT_EQ(bridge.state().loaders.load(), 0); +} + +} // namespace +} // namespace executorch::backends::coreai::testing diff --git a/backends/apple/coreai/runtime/test/coreai_fake_loader.h b/backends/apple/coreai/runtime/test/coreai_fake_loader.h index 7732fee3507..95e2f8e371d 100644 --- a/backends/apple/coreai/runtime/test/coreai_fake_loader.h +++ b/backends/apple/coreai/runtime/test/coreai_fake_loader.h @@ -20,11 +20,28 @@ namespace executorch::backends::coreai::testing { // Configure on the test thread before starting workers; join before changing // it. struct FakeBridgeState { + bool available = true; + bool fail_loader_factory = false; + NSString* device_architecture = @"arch_b"; id bookmark_loader = nil; + bool fail_load = false; + bool fail_execute = false; + bool bad_output = false; + NSString* last_bundle = nil; + dispatch_semaphore_t input_ready = nil; + dispatch_semaphore_t allow_input_read = nil; + std::atomic architecture_queries{0}; + std::atomic executions{0}; + std::atomic sessions{0}; std::atomic prepared_models{0}; std::atomic loaders{0}; + std::atomic last_input_bytes{nullptr}; + std::atomic last_input_byte_count{0}; + std::atomic last_input_first_byte{0}; // Worker-side contract failures are checked by the scope on the test thread. + std::atomic binding_mismatches{0}; std::atomic missing_bundles{0}; + std::atomic input_wait_timeouts{0}; }; // Process-visible, not thread-local. Scopes and configuration changes are @@ -34,7 +51,7 @@ class ScopedFakeBridgeState final { ScopedFakeBridgeState(); ~ScopedFakeBridgeState(); FakeBridgeState& state(); - // Join caller threads before draining callbacks. + // Release input gates and join caller threads before draining callbacks. ::testing::AssertionResult wait_for_callbacks(dispatch_time_t deadline); ScopedFakeBridgeState(const ScopedFakeBridgeState&) = delete; ScopedFakeBridgeState& operator=(const ScopedFakeBridgeState&) = delete; @@ -46,6 +63,17 @@ class ScopedFakeBridgeState final { } // namespace executorch::backends::coreai::testing +@interface FakeSession : NSObject +@property(nonatomic, strong) id preparedPin; +@end + +@interface FakePreparedModel : NSObject +@property(nonatomic, copy) NSData* bookmark; +@end + +@interface FakeLoader : NSObject +@end + @interface BookmarkFakeModel : NSObject @property(nonatomic, weak) BookmarkFakeLoader* owner; @property(nonatomic, copy) NSData* bookmark; @@ -55,12 +83,14 @@ class ScopedFakeBridgeState final { @public std::atomic restores; std::atomic specializations; + std::atomic bindings; std::atomic evictions; } @property(nonatomic, strong) NSMutableDictionary* models; @property(nonatomic, strong) NSError* restoreError; @property(nonatomic, strong) NSError* specializeError; +@property(nonatomic, strong) NSError* bindError; @property(nonatomic, strong) NSError* evictError; @property(nonatomic) BOOL restoreMiss; @property(nonatomic) BOOL emptyBookmark; @@ -68,6 +98,7 @@ class ScopedFakeBridgeState final { @property(nonatomic, copy) void (^onSpecialize)(NSURL*); @property(nonatomic, copy) void (^onEvict)(NSData*); @property(nonatomic, strong) BookmarkFakeModel* lastAcquired; +@property(nonatomic, strong) BookmarkFakeModel* lastBound; @property(nonatomic, copy) NSString* lastSpecializeURL; // Call only after workers/callbacks finish and caller-owned handles are // released. diff --git a/backends/apple/coreai/runtime/test/coreai_fake_loader.mm b/backends/apple/coreai/runtime/test/coreai_fake_loader.mm index a997ba361fa..38153112675 100644 --- a/backends/apple/coreai/runtime/test/coreai_fake_loader.mm +++ b/backends/apple/coreai/runtime/test/coreai_fake_loader.mm @@ -34,6 +34,17 @@ - (instancetype)init { return [NSError errorWithDomain:ETCoreAIErrorDomain code:code userInfo:nil]; } +bool valid_binding(CoreAIFakeBridgeContext* context, NSString* function, NSArray* inputs, + NSArray* outputs) { + const bool valid = [function isEqualToString:@"main"] && + [inputs isEqualToArray:@[ @"input_0" ]] && + [outputs isEqualToArray:@[ @"output_0" ]]; + if (!valid) { + ++context->state.binding_mismatches; + } + return valid; +} + // Release callback captures before publishing that all fake work has drained. void run_async(CoreAIFakeBridgeContext* context, void (^work)()) { dispatch_group_t group = context.callbacks; @@ -59,6 +70,159 @@ void run_async(CoreAIFakeBridgeContext* context, void (^work)()) { dispatch_time_t cleanup_deadline() { return dispatch_time(DISPATCH_TIME_NOW, 10 * NSEC_PER_SEC); } } // namespace +@interface FakeSession () +@property(nonatomic, strong) CoreAIFakeBridgeContext* context; +- (instancetype)initWithContext:(CoreAIFakeBridgeContext*)context; +@end + +@implementation FakeSession +- (instancetype)init { + return [self initWithContext:current_context]; +} +- (instancetype)initWithContext:(CoreAIFakeBridgeContext*)context { + if (context == nil) { + return nil; + } + if ((self = [super init])) { + self.context = context; + ++context->state.sessions; + } + return self; +} +- (void)dealloc { + if (_context != nil) { + --_context->state.sessions; + } +} +- (void)executeInputs:(NSArray*)inputs + completion:(void (^)(NSArray*, NSError*))completion { + CoreAIFakeBridgeContext* context = self.context; + ++context->state.executions; + const bool should_fail = context->state.fail_execute; + const bool malformed = context->state.bad_output; + dispatch_semaphore_t ready = context->state.input_ready; + dispatch_semaphore_t proceed = context->state.allow_input_read; + run_async(context, ^{ + ETCoreAIInputTensor* input = inputs[0]; + context->state.last_input_bytes = input.bytes; + context->state.last_input_byte_count = input.byteCount; + if (ready != nil) { + dispatch_semaphore_signal(ready); + if (proceed == nil || dispatch_semaphore_wait(proceed, cleanup_deadline()) != 0) { + ++context->state.input_wait_timeouts; + completion(nil, fake_error(ETCoreAIErrorRuntime)); + return; + } + } + if (input.byteCount > 0) { + context->state.last_input_first_byte = static_cast(input.bytes)[0]; + } + if (should_fail) { + completion(nil, fake_error(ETCoreAIErrorRuntime)); + return; + } + if (malformed) { + completion(@[ [[ETCoreAITensor alloc] initWithData:[NSData data] + shape:@[ @2 ] + scalarType:ETCoreAIScalarTypeFloat32] ], + nil); + return; + } + NSData* data = [NSData dataWithBytes:input.bytes length:input.byteCount]; + ETCoreAITensor* output = [[ETCoreAITensor alloc] initWithData:data + shape:input.shape + scalarType:input.scalarType]; + completion(@[ output ], nil); + }); +} +@end + +@interface FakePreparedModel () +@property(nonatomic, strong) CoreAIFakeBridgeContext* context; +- (instancetype)initWithContext:(CoreAIFakeBridgeContext*)context; +@end + +@implementation FakePreparedModel +- (instancetype)init { + return [self initWithContext:current_context]; +} +- (instancetype)initWithContext:(CoreAIFakeBridgeContext*)context { + if (context == nil) { + return nil; + } + if ((self = [super init])) { + self.context = context; + ++context->state.prepared_models; + } + return self; +} +- (void)dealloc { + if (_context != nil) { + --_context->state.prepared_models; + } +} +- (NSData*)copyBookmarkData { + return self.bookmark; +} +- (void)loadFunctionNamed:(NSString*)functionName + inputNames:(NSArray*)inputNames + outputNames:(NSArray*)outputNames + completion:(void (^)(id, NSError*))completion { + CoreAIFakeBridgeContext* context = self.context; + if (!valid_binding(context, functionName, inputNames, outputNames) || context->state.fail_load) { + completion(nil, fake_error(ETCoreAIErrorInvalidModel)); + return; + } + FakeSession* session = [[FakeSession alloc] initWithContext:context]; + session.preparedPin = self; + completion(session, nil); +} +@end + +@interface FakeLoader () +@property(nonatomic, strong) CoreAIFakeBridgeContext* context; +@end + +@implementation FakeLoader +- (instancetype)init { + if (current_context == nil) { + return nil; + } + if ((self = [super init])) { + self.context = current_context; + ++_context->state.loaders; + } + return self; +} +- (void)dealloc { + if (_context != nil) { + --_context->state.loaders; + } +} +- (void)restoreModelFromBookmark:(NSData*)bookmark + completion:(ETCoreAIAcquisitionCompletion)completion { + FakePreparedModel* model = [[FakePreparedModel alloc] initWithContext:self.context]; + model.bookmark = bookmark; + _context->state.last_bundle = [[NSString alloc] initWithData:bookmark + encoding:NSUTF8StringEncoding]; + completion(ETCoreAIAcquisitionStatusHit, model, nil); +} +- (void)specializeModelAtURL:(NSURL*)url completion:(ETCoreAIAcquisitionCompletion)completion { + if (![NSFileManager.defaultManager fileExistsAtPath:url.path]) { + ++_context->state.missing_bundles; + completion(ETCoreAIAcquisitionStatusError, nil, fake_error(ETCoreAIErrorInvalidModel)); + return; + } + _context->state.last_bundle = url.path; + FakePreparedModel* model = [[FakePreparedModel alloc] initWithContext:self.context]; + model.bookmark = [url.path dataUsingEncoding:NSUTF8StringEncoding]; + completion(ETCoreAIAcquisitionStatusHit, model, nil); +} +- (void)evictModelWithBookmark:(NSData*)bookmark completion:(void (^)(NSError*))completion { + completion(nil); +} +@end + @interface BookmarkFakeModel () @property(nonatomic, strong) CoreAIFakeBridgeContext* context; - (instancetype)initWithContext:(CoreAIFakeBridgeContext*)context; @@ -94,8 +258,23 @@ - (void)loadFunctionNamed:(NSString*)functionName inputNames:(NSArray*)inputNames outputNames:(NSArray*)outputNames completion:(void (^)(id, NSError*))completion { - // Acquisition tests never bind functions. - completion(nil, fake_error(ETCoreAIErrorUnsupported)); + CoreAIFakeBridgeContext* context = self.context; + BookmarkFakeLoader* owner = self.owner; + if (!valid_binding(context, functionName, inputNames, outputNames) || owner == nil) { + completion(nil, fake_error(ETCoreAIErrorInvalidModel)); + return; + } + ++owner->bindings; + owner.lastBound = self; + run_async(context, ^{ + if (owner.bindError != nil) { + completion(nil, owner.bindError); + return; + } + FakeSession* session = [[FakeSession alloc] initWithContext:context]; + session.preparedPin = self; + completion(session, nil); + }); } @end @@ -109,6 +288,7 @@ - (instancetype)init { self.models = [NSMutableDictionary dictionary]; restores = 0; specializations = 0; + bindings = 0; evictions = 0; ++_context->state.loaders; } @@ -181,6 +361,7 @@ - (void)clearRetainedState { self.onSpecialize = nil; self.onEvict = nil; self.lastAcquired = nil; + self.lastBound = nil; [self.models removeAllObjects]; } @end @@ -202,9 +383,14 @@ - (void)clearRetainedState { } context_->state.bookmark_loader = nil; loader = nil; + context_->state.input_ready = nil; + context_->state.allow_input_read = nil; + EXPECT_EQ(context_->state.sessions.load(), 0); EXPECT_EQ(context_->state.prepared_models.load(), 0); EXPECT_EQ(context_->state.loaders.load(), 0); + EXPECT_EQ(context_->state.binding_mismatches.load(), 0); EXPECT_EQ(context_->state.missing_bundles.load(), 0); + EXPECT_EQ(context_->state.input_wait_timeouts.load(), 0); } EXPECT_EQ(current_context, context_); current_context = previous_; @@ -241,3 +427,22 @@ - (void)clearRetainedState { } } // namespace executorch::backends::coreai::testing + +BOOL ETCoreAIIsAvailable(void) { + return current_context != nil && current_context->state.available; +} + +NSString* ETCoreAIDeviceArchitectureName(void) { + if (current_context == nil) { + return nil; + } + ++current_context->state.architecture_queries; + return current_context->state.device_architecture; +} + +id ETCoreAICreateModelLoader(void) { + if (current_context == nil || current_context->state.fail_loader_factory) { + return nil; + } + return current_context->state.bookmark_loader ?: [[FakeLoader alloc] init]; +} diff --git a/backends/apple/coreai/runtime/test/runtime_smoke.cpp b/backends/apple/coreai/runtime/test/runtime_smoke.cpp new file mode 100644 index 00000000000..3d91a760b4b --- /dev/null +++ b/backends/apple/coreai/runtime/test/runtime_smoke.cpp @@ -0,0 +1,27 @@ +/* + * Copyright (c) Meta Platforms, Inc. and affiliates. + * All rights reserved. + * + * This source code is licensed under the BSD-style license found in the + * LICENSE file in the root directory of this source tree. + */ + +#include +#include + +#include + +int main() { + executorch::runtime::runtime_init(); + const auto* backend = executorch::runtime::get_backend_class("CoreAIBackend"); + if (backend == nullptr) { + std::fputs("CoreAIBackend was not retained by the linker\n", stderr); + return 1; + } + if (!backend->is_available()) { + std::fputs("CoreAIBackend requires OS 27 or newer\n", stderr); + return 1; + } + std::puts("CoreAIBackend is registered and available"); + return 0; +} diff --git a/tools/cmake/preset/default.cmake b/tools/cmake/preset/default.cmake index 81c5c2b3f97..3411404ac72 100644 --- a/tools/cmake/preset/default.cmake +++ b/tools/cmake/preset/default.cmake @@ -416,6 +416,10 @@ check_required_options_on( EXECUTORCH_BUILD_EXTENSION_LLM ) +check_required_options_on( + IF_ON EXECUTORCH_BUILD_COREAI REQUIRES EXECUTORCH_BUILD_EXTENSION_DATA_LOADER +) + check_required_options_on( IF_ON EXECUTORCH_BUILD_EXTENSION_MODULE REQUIRES EXECUTORCH_BUILD_EXTENSION_DATA_LOADER EXECUTORCH_BUILD_EXTENSION_FLAT_TENSOR