From 945d65faf86a70f55afa8645d5495c762ccfe095 Mon Sep 17 00:00:00 2001 From: Baiju Meswani Date: Sun, 13 Sep 2026 08:28:20 +0000 Subject: [PATCH 1/5] Update to ort 1.3.0.0 and genai to 0.16.0 --- .../cpp-ci-test-policy.instructions.md | 6 +- .pipelines/foundry-local-packaging.yml | 15 ++--- .pipelines/v2/sdk_v2-pipeline-plan.md | 5 +- .pipelines/v2/templates/stages-sdk-v2.yml | 3 +- .../v2/templates/stages-test-engine.yml | 6 +- sdk_v2/cpp/CMakeLists.txt | 61 ++----------------- .../cpp/src/ep_detection/cuda_ep_manifest.cc | 30 ++++----- .../generative/chat/chat_session.cc | 7 --- .../generative/genai_model_instance.cc | 6 -- .../generative/genai_model_instance.h | 6 -- sdk_v2/cpp/test/CMakeLists.txt | 12 +--- .../internal_api/cuda_ep_bootstrapper_test.cc | 36 +++++------ sdk_v2/deps_versions.json | 4 +- sdk_v2/js/test/install-native.test.ts | 32 +++++----- sdk_v2/rust/deps_versions.json | 4 +- 15 files changed, 73 insertions(+), 160 deletions(-) diff --git a/.github/instructions/cpp-ci-test-policy.instructions.md b/.github/instructions/cpp-ci-test-policy.instructions.md index e70ac03b9..dca4f0b9b 100644 --- a/.github/instructions/cpp-ci-test-policy.instructions.md +++ b/.github/instructions/cpp-ci-test-policy.instructions.md @@ -34,11 +34,9 @@ unload, not CUDA kernel correctness or trained-model quality. The packaging pipeline includes the unconditional `cpp_test_engine` stage from `.pipelines/v2/templates/stages-test-engine.yml`. It runs on the existing `onnxruntime-Ubuntu2404-AMD-CPU` pool image -without a custom container and uses standard CPU NuGet packages with a test-only Engine-capable GenAI pin. -`FOUNDRY_LOCAL_REQUIRE_DYNAMIC_ENGINE_TESTS=ON` rejects packages that cannot compile the suite. The XML gate rejects +without a custom container and uses the same stable ORT and GenAI packages as the SDK build. The XML gate rejects missing lifecycle cases, skipped/disabled/unexecuted tests, and failures. Do not bypass failures with skips or -`continueOnError`. Its binaries are not published as SDK artifacts, and shipping/release dependency pins are unchanged. -Generator-only builds can still exclude the suite; the unconditional Engine lane supplies the required coverage. +`continueOnError`. Its binaries are not published as SDK artifacts. ## Debugging skips in CI diff --git a/.pipelines/foundry-local-packaging.yml b/.pipelines/foundry-local-packaging.yml index 2ee3d2e0b..51d9b2365 100644 --- a/.pipelines/foundry-local-packaging.yml +++ b/.pipelines/foundry-local-packaging.yml @@ -53,17 +53,12 @@ parameters: variables: - group: FoundryLocal-ESRP-Signing -# C++ SDK (sdk_v2/cpp) native dependency versions. Release builds use the -# stable pins from sdk_v2/deps_versions.json; non-release CI validates the -# selected GenAI nightly before it becomes a stable SDK default. +# C++ SDK (sdk_v2/cpp) native dependency versions. Must match the stable pins +# in sdk_v2/deps_versions.json. - name: cppOrtVersion - value: '1.28.0' -- ${{ if eq(parameters.isRelease, true) }}: - - name: cppGenaiVersion - value: '0.15.2' -- ${{ else }}: - - name: cppGenaiVersion - value: '0.16.0-dev1001400138' + value: '1.30.0' +- name: cppGenaiVersion + value: '0.16.0' - name: cppWinmlVersion value: '2.1.70' - name: cppBuildConfig diff --git a/.pipelines/v2/sdk_v2-pipeline-plan.md b/.pipelines/v2/sdk_v2-pipeline-plan.md index 44fc170f6..c3809c649 100644 --- a/.pipelines/v2/sdk_v2-pipeline-plan.md +++ b/.pipelines/v2/sdk_v2-pipeline-plan.md @@ -284,9 +284,8 @@ purposes: Versions are pipeline-level variables, currently: -* `ortVersion` `1.28.0` (`Microsoft.ML.OnnxRuntime`) -* `genaiVersion` `0.15.2` for releases; selected ORT-Nightly version for non-release CI - (`Microsoft.ML.OnnxRuntimeGenAI.Foundry`) +* `ortVersion` `1.30.0` (`Microsoft.ML.OnnxRuntime`) +* `genaiVersion` `0.16.0` (`Microsoft.ML.OnnxRuntimeGenAI.Foundry`) * `winmlVersion` `2.1.70` (`Microsoft.Windows.AI.MachineLearning`, WinML 2.x reg-free) These must be kept in sync with the cmake defaults and with diff --git a/.pipelines/v2/templates/stages-sdk-v2.yml b/.pipelines/v2/templates/stages-sdk-v2.yml index a59f9b961..b9e17ea99 100644 --- a/.pipelines/v2/templates/stages-sdk-v2.yml +++ b/.pipelines/v2/templates/stages-sdk-v2.yml @@ -33,11 +33,12 @@ stages: genaiVersion: ${{ parameters.genaiVersion }} winmlVersion: ${{ parameters.winmlVersion }} -# Test-only CUDA runtime; never replaces the native packaging artifacts. +# Dedicated Engine lifecycle test stage. - template: stages-test-engine.yml parameters: buildConfig: ${{ parameters.buildConfig }} ortVersion: ${{ parameters.ortVersion }} + genaiVersion: ${{ parameters.genaiVersion }} # ── C# SDK ── - template: stages-cs.yml diff --git a/.pipelines/v2/templates/stages-test-engine.yml b/.pipelines/v2/templates/stages-test-engine.yml index d54c2c429..597492c08 100644 --- a/.pipelines/v2/templates/stages-test-engine.yml +++ b/.pipelines/v2/templates/stages-test-engine.yml @@ -1,4 +1,4 @@ -# CPU paged-attention Engine gate, independent of the Generator/release packaging matrix. +# CPU paged-attention Engine gate. # Model assets are checked in and staged by the normal C++ testdata build rule. parameters: - name: buildConfig @@ -7,7 +7,6 @@ parameters: type: string - name: genaiVersion type: string - default: '0.16.0-dev1001407373' stages: - stage: cpp_test_engine @@ -48,8 +47,7 @@ stages: --cmake_extra_defines \ "GENAI_FETCH_URL=$NUGET_CACHE/genai.zip" \ "ORT_GENAI_VERSION=$ENGINE_GENAI_VERSION" \ - "ORT_FETCH_URL=$NUGET_CACHE/ort.zip" \ - "FOUNDRY_LOCAL_REQUIRE_DYNAMIC_ENGINE_TESTS=ON" + "ORT_FETCH_URL=$NUGET_CACHE/ort.zip" build="$PWD/build/Linux/$BUILD_CONFIG" bin="$build/bin" export LD_LIBRARY_PATH="$bin:${LD_LIBRARY_PATH:-}" diff --git a/sdk_v2/cpp/CMakeLists.txt b/sdk_v2/cpp/CMakeLists.txt index 00a14a20d..3bdbe67df 100644 --- a/sdk_v2/cpp/CMakeLists.txt +++ b/sdk_v2/cpp/CMakeLists.txt @@ -50,12 +50,6 @@ option(FOUNDRY_LOCAL_BUILD_TESTS "Build unit tests" ON) option(FOUNDRY_LOCAL_BUILD_EXAMPLES "Build example programs" ON) option(FOUNDRY_LOCAL_BUILD_SERVICE "Build web service support (requires oat++)" ON) option(FOUNDRY_LOCAL_ENABLE_ASAN "Enable AddressSanitizer + UndefinedBehaviorSanitizer (Linux only)" OFF) -option(FOUNDRY_LOCAL_REQUIRE_DYNAMIC_ENGINE_TESTS - "Fail configuration unless the ORT GenAI dynamic Engine tests can be built" - OFF) -if(FOUNDRY_LOCAL_REQUIRE_DYNAMIC_ENGINE_TESTS AND NOT FOUNDRY_LOCAL_BUILD_TESTS) - message(FATAL_ERROR "FOUNDRY_LOCAL_REQUIRE_DYNAMIC_ENGINE_TESTS requires FOUNDRY_LOCAL_BUILD_TESTS=ON") -endif() # Optional 1DS ingestion token override. Override only via environment so it does # not appear in CMake cache files. @@ -115,50 +109,6 @@ endif() # ORT and ORT GenAI — acquired via FetchContent from nuget.org. list(APPEND CMAKE_MODULE_PATH "${CMAKE_CURRENT_SOURCE_DIR}/cmake") find_package(OnnxRuntimeGenAI REQUIRED) - -include(CheckCXXSourceCompiles) - -set(CMAKE_REQUIRED_LIBRARIES OnnxRuntimeGenAI::OnnxRuntimeGenAI) -unset(FOUNDRY_LOCAL_OGA_HAS_DYNAMIC_ENGINE CACHE) -check_cxx_source_compiles(" -#include -#include -#include - -#include - -int main() { - auto* model = reinterpret_cast(1); - auto engine = OgaEngine::Create(*model); - auto events = engine->CreateEventBuffer(1); - auto request_options = OgaRequestOptions::Create(); - auto request = engine->CreateRequest(request_options.get()); - auto turn_options = request->CreateTurnOptions(); - turn_options->SetMaxGeneratedTokens(static_cast(1)); - turn_options->SetDoSample(true); - turn_options->SetTemperature(1.0f); - turn_options->SetTopP(1.0f); - turn_options->SetTopK(1); - turn_options->SetSeed(static_cast(1)); - const char* values[] = {\"stop\"}; - auto stop_strings = OgaStringArray::Create(values, 1); - turn_options->SetStopStrings(*stop_strings); - turn_options->SetGuidance(\"\", \"\"); - const auto finish_reason = OgaFinishReason_StopString; - return events && request && turn_options && finish_reason == OgaFinishReason_Eos ? 0 : 1; -} -" FOUNDRY_LOCAL_OGA_HAS_DYNAMIC_ENGINE) -unset(CMAKE_REQUIRED_LIBRARIES) - -if(FOUNDRY_LOCAL_OGA_HAS_DYNAMIC_ENGINE) - message(STATUS "ORT GenAI dynamic Engine API: enabled") -else() - if(FOUNDRY_LOCAL_REQUIRE_DYNAMIC_ENGINE_TESTS) - message(FATAL_ERROR - "FOUNDRY_LOCAL_REQUIRE_DYNAMIC_ENGINE_TESTS requires an ORT GenAI package with the dynamic Engine API") - endif() - message(STATUS "ORT GenAI dynamic Engine API: unavailable; using Generator backend") -endif() find_package(OnnxRuntime REQUIRED) # WinML EP Catalog — Windows-only, for hardware EP discovery and download. The @@ -331,12 +281,10 @@ set(FOUNDRY_LOCAL_SOURCES ${FOUNDRY_LOCAL_INTERNAL_HEADERS} ) -if(FOUNDRY_LOCAL_OGA_HAS_DYNAMIC_ENGINE) - list(APPEND FOUNDRY_LOCAL_SOURCES - src/inferencing/generative/chat/onnx_chat_engine.cc - src/inferencing/generative/chat/onnx_engine_chat_stream.cc - ) -endif() +list(APPEND FOUNDRY_LOCAL_SOURCES + src/inferencing/generative/chat/onnx_chat_engine.cc + src/inferencing/generative/chat/onnx_engine_chat_stream.cc +) # 1DS bridge — always compiled for Foundry Local Core. list(APPEND FOUNDRY_LOCAL_SOURCES src/telemetry/one_ds_telemetry.cc) @@ -412,7 +360,6 @@ function(foundry_local_configure_target TARGET LINK_SCOPE) ${TARGET} PRIVATE FOUNDRY_LOCAL_HAS_EP_BOOTSTRAPPERS=$ - FOUNDRY_LOCAL_OGA_HAS_DYNAMIC_ENGINE=$ ) if(FOUNDRY_LOCAL_BUILD_SERVICE) diff --git a/sdk_v2/cpp/src/ep_detection/cuda_ep_manifest.cc b/sdk_v2/cpp/src/ep_detection/cuda_ep_manifest.cc index 1c509beaf..1cb46df85 100644 --- a/sdk_v2/cpp/src/ep_detection/cuda_ep_manifest.cc +++ b/sdk_v2/cpp/src/ep_detection/cuda_ep_manifest.cc @@ -32,7 +32,7 @@ EpBundleArtifact Archive(std::string id, std::string filename, std::string sha25 EpBundleManifest WindowsX64Manifest() { return EpBundleManifest{ - .bundle_id = "cuda-ep-win-x64-cuda-12.8.4-ort-1.28.0-genai-0.15.2-20260806-182620", + .bundle_id = "cuda-ep-win-x64-cuda-12.8.4-ort-1.30.0-genai-0.16.0-20260913-071149", .artifacts = { Archive( @@ -66,13 +66,13 @@ EpBundleManifest WindowsX64Manifest() { .sha256 = "49487537744256a3d4365c4792b03bf31130ad1faea0a13eafa219620941d837"}, }), Archive( - "cuda-ep", "cuda-ep-bins-win-x64-20260806-182620.zip", - "e62938987e848a0fbb3d215dfefaed40307d2446393909927ba0345eaaf3d263", 256 * kMiB, + "cuda-ep", "cuda-ep-bins-win-x64-20260913-071149.zip", + "5d1c38eb4058b6898d78ae5e7881ac07dda8ef37156be232f0f63846a1405f85", 256 * kMiB, { {.relative_path = "onnxruntime-genai-cuda.dll", - .sha256 = "7894fb5efaad4a663e834f20b912b44cc383629b24ffe8bbc6382786a7326dbc"}, + .sha256 = "132d28f988ee8bc1ce050e8dcdda7036ab436616d827508af8e0a4f6ec21f2da"}, {.relative_path = "onnxruntime_providers_cuda.dll", - .sha256 = "60f1aeef7ebe27f7e659cb88f597005ca5a5e75832b85dcef3eef02b9322df9a"}, + .sha256 = "7d014892b64d03092c01e99ae5786c6080c5239eeee30e6fb902fd2a800fa95e"}, }), }, .provider_relative_path = "onnxruntime_providers_cuda.dll", @@ -81,7 +81,7 @@ EpBundleManifest WindowsX64Manifest() { EpBundleManifest WindowsArm64Manifest() { return EpBundleManifest{ - .bundle_id = "cuda-ep-win-arm64-cuda-13.4.1-ort-1.28.0-genai-0.15.2-20260806-182803", + .bundle_id = "cuda-ep-win-arm64-cuda-13.4.1-ort-1.30.0-genai-0.16.0-20260913-071350", .artifacts = { Archive( @@ -115,13 +115,13 @@ EpBundleManifest WindowsArm64Manifest() { .sha256 = "c9e0ec0e0a4e659393e15897ed1f6e5bac677e0c0fe7e12290f0386f19477b6b"}, }), Archive( - "cuda-ep", "cuda-ep-bins-win-arm64-20260806-182803.zip", - "212e670c61b3292d4a7d98f16fc2cf61f7b080604e0c145e81c39ec81e7b3259", 96 * kMiB, + "cuda-ep", "cuda-ep-bins-win-arm64-20260913-071350.zip", + "fc8f0a01daafedc82aa57072f023b11cf801dd94ed0a449ddc8ac4698e43e519", 96 * kMiB, { {.relative_path = "onnxruntime-genai-cuda.dll", - .sha256 = "ab61145f4bc6284286e663586f634b973072d58ced20c497c7e5259f2ef3fc08"}, + .sha256 = "d27a02b8a83d0aaac47904d1baff4f7954aa25dcc130dba539fe9aafc53eab30"}, {.relative_path = "onnxruntime_providers_cuda.dll", - .sha256 = "d92ffbd23a84f91b976baed9031de267efe1dc892d85c09d0979d25b89f5d1a0"}, + .sha256 = "82a3887c64791fc7131a0f705130f02f456952915ecea71372236656c59a47b2"}, }), }, .provider_relative_path = "onnxruntime_providers_cuda.dll", @@ -130,17 +130,17 @@ EpBundleManifest WindowsArm64Manifest() { EpBundleManifest LinuxX64Manifest() { return EpBundleManifest{ - .bundle_id = "cuda-ep-linux-x64-ort-1.28.0-genai-0.15.2-20260806-182830", + .bundle_id = "cuda-ep-linux-x64-ort-1.30.0-genai-0.16.0-20260913-071417", .artifacts = { Archive( - "cuda-ep", "cuda-ep-linux-x64-20260806-182830.zip", - "abf347e7234d7434105efde12a2e0609fdd1d8828167b9873f4463926f1206e6", 448 * kMiB, + "cuda-ep", "cuda-ep-linux-x64-20260913-071417.zip", + "482e615e4c66b3f14a980af00e330e28fa5897a34161ec2972f75289685546be", 448 * kMiB, { {.relative_path = "libonnxruntime-genai-cuda.so", - .sha256 = "d5300fc4413d9e74bd8dfceb5233fca6fcfa1d5ddc247081365fdb5f143091e6"}, + .sha256 = "3c95bb3b92f560718482dd3638388b6fd7374fb5ba6d9d5f4c4ca22c095e7681"}, {.relative_path = "libonnxruntime_providers_cuda.so", - .sha256 = "b88d7b7f4b2e81d3eff41663fc70f4ae9e03dee9e2301cb53dc250e5a96d7f7a"}, + .sha256 = "8589ae31ed4941e72693a6a2e9dd8b8e721c888cfd9e6ffcb26da4e6e58ce720"}, }), }, .provider_relative_path = "libonnxruntime_providers_cuda.so", diff --git a/sdk_v2/cpp/src/inferencing/generative/chat/chat_session.cc b/sdk_v2/cpp/src/inferencing/generative/chat/chat_session.cc index e5264bccf..d7565aee0 100644 --- a/sdk_v2/cpp/src/inferencing/generative/chat/chat_session.cc +++ b/sdk_v2/cpp/src/inferencing/generative/chat/chat_session.cc @@ -8,9 +8,7 @@ #include "inferencing/generative/chat/media_input.h" #include "inferencing/generative/chat/onnx_chat_engine.h" #include "inferencing/generative/chat/onnx_chat_generator.h" -#if FOUNDRY_LOCAL_OGA_HAS_DYNAMIC_ENGINE #include "inferencing/generative/chat/onnx_engine_chat_stream.h" -#endif #include "inferencing/generative/chat/reasoning_stream_splitter.h" #include "inferencing/generative/chat/stop_strings.h" #include "inferencing/generative/genai_model_instance.h" @@ -61,12 +59,7 @@ std::unique_ptr CreateTextChatGenerator(const std::vector(*this); } catch (const std::runtime_error& e) { FL_LOG_AND_THROW(logger, FOUNDRY_LOCAL_ERROR_INTERNAL, "failed to create chat engine for model ", model_id_, ": ", e.what()); } -#else - FL_LOG_AND_THROW(logger, FOUNDRY_LOCAL_ERROR_INTERNAL, - "model ", model_id_, - " requires the ORT GenAI dynamic Engine API, but this build does not provide it"); -#endif } } diff --git a/sdk_v2/cpp/src/inferencing/generative/genai_model_instance.h b/sdk_v2/cpp/src/inferencing/generative/genai_model_instance.h index 17ecacfc5..ed273599a 100644 --- a/sdk_v2/cpp/src/inferencing/generative/genai_model_instance.h +++ b/sdk_v2/cpp/src/inferencing/generative/genai_model_instance.h @@ -62,11 +62,7 @@ class GenAIModelInstance { /// Access the underlying OGA objects. OgaModel& GetOgaModel(); Preprocessor& GetPreprocessor(); -#if FOUNDRY_LOCAL_OGA_HAS_DYNAMIC_ENGINE OnnxChatEngine* GetChatEngine() { return chat_engine_.get(); } -#else - OnnxChatEngine* GetChatEngine() { return nullptr; } -#endif /// Get the last-activity timestamp. std::chrono::steady_clock::time_point LastActivity() const { return last_activity_; } @@ -93,9 +89,7 @@ class GenAIModelInstance { ExecutionProvider ep_; std::unique_ptr oga_model_; std::unique_ptr preprocessor_; -#if FOUNDRY_LOCAL_OGA_HAS_DYNAMIC_ENGINE std::unique_ptr chat_engine_; -#endif TagInfo tag_info_; std::once_flag tag_info_init_flag_; std::chrono::steady_clock::time_point last_activity_; diff --git a/sdk_v2/cpp/test/CMakeLists.txt b/sdk_v2/cpp/test/CMakeLists.txt index 4b3e9bc06..112086158 100644 --- a/sdk_v2/cpp/test/CMakeLists.txt +++ b/sdk_v2/cpp/test/CMakeLists.txt @@ -95,17 +95,11 @@ if(FOUNDRY_LOCAL_HAS_EP_BOOTSTRAPPERS) ) endif() -if(FOUNDRY_LOCAL_OGA_HAS_DYNAMIC_ENGINE) - target_sources(foundry_local_tests PRIVATE - internal_api/chat/dynamic_engine_chat_test.cc - ) -endif() +target_sources(foundry_local_tests PRIVATE + internal_api/chat/dynamic_engine_chat_test.cc +) target_compile_options(foundry_local_tests PRIVATE ${FOUNDRY_LOCAL_COMPILE_OPTIONS}) -target_compile_definitions( - foundry_local_tests - PRIVATE FOUNDRY_LOCAL_OGA_HAS_DYNAMIC_ENGINE=$ -) target_link_libraries(foundry_local_tests PRIVATE diff --git a/sdk_v2/cpp/test/internal_api/cuda_ep_bootstrapper_test.cc b/sdk_v2/cpp/test/internal_api/cuda_ep_bootstrapper_test.cc index 3d820a329..db1c69410 100644 --- a/sdk_v2/cpp/test/internal_api/cuda_ep_bootstrapper_test.cc +++ b/sdk_v2/cpp/test/internal_api/cuda_ep_bootstrapper_test.cc @@ -151,10 +151,10 @@ TEST(CudaEpBootstrapperTest, PlatformSupportMatchesPublishedBundles) { #endif } -TEST(CudaEpManifestTest, WindowsX64MetadataMatches20260806Bundle) { +TEST(CudaEpManifestTest, WindowsX64MetadataMatches20260913Bundle) { const auto manifest = BuildCudaEpManifest(CudaEpPlatform::WindowsX64); ASSERT_TRUE(manifest.has_value()); - EXPECT_EQ(manifest->bundle_id, "cuda-ep-win-x64-cuda-12.8.4-ort-1.28.0-genai-0.15.2-20260806-182620"); + EXPECT_EQ(manifest->bundle_id, "cuda-ep-win-x64-cuda-12.8.4-ort-1.30.0-genai-0.16.0-20260913-071149"); EXPECT_EQ(manifest->provider_relative_path, "onnxruntime_providers_cuda.dll"); ASSERT_EQ(manifest->artifacts.size(), 3u); @@ -179,19 +179,19 @@ TEST(CudaEpManifestTest, WindowsX64MetadataMatches20260806Bundle) { {"cudnn_ops64_9.dll", "49487537744256a3d4365c4792b03bf31130ad1faea0a13eafa219620941d837"}, }); ExpectArtifact( - manifest->artifacts[2], "cuda-ep", "cuda-ep-bins-win-x64-20260806-182620.zip", - "e62938987e848a0fbb3d215dfefaed40307d2446393909927ba0345eaaf3d263", 256 * kMiB, + manifest->artifacts[2], "cuda-ep", "cuda-ep-bins-win-x64-20260913-071149.zip", + "5d1c38eb4058b6898d78ae5e7881ac07dda8ef37156be232f0f63846a1405f85", 256 * kMiB, { - {"onnxruntime-genai-cuda.dll", "7894fb5efaad4a663e834f20b912b44cc383629b24ffe8bbc6382786a7326dbc"}, - {"onnxruntime_providers_cuda.dll", "60f1aeef7ebe27f7e659cb88f597005ca5a5e75832b85dcef3eef02b9322df9a"}, + {"onnxruntime-genai-cuda.dll", "132d28f988ee8bc1ce050e8dcdda7036ab436616d827508af8e0a4f6ec21f2da"}, + {"onnxruntime_providers_cuda.dll", "7d014892b64d03092c01e99ae5786c6080c5239eeee30e6fb902fd2a800fa95e"}, }); ExpectUniqueInstalledPaths(*manifest); } -TEST(CudaEpManifestTest, WindowsArm64MetadataMatches20260806Bundle) { +TEST(CudaEpManifestTest, WindowsArm64MetadataMatches20260913Bundle) { const auto manifest = BuildCudaEpManifest(CudaEpPlatform::WindowsArm64); ASSERT_TRUE(manifest.has_value()); - EXPECT_EQ(manifest->bundle_id, "cuda-ep-win-arm64-cuda-13.4.1-ort-1.28.0-genai-0.15.2-20260806-182803"); + EXPECT_EQ(manifest->bundle_id, "cuda-ep-win-arm64-cuda-13.4.1-ort-1.30.0-genai-0.16.0-20260913-071350"); EXPECT_EQ(manifest->provider_relative_path, "onnxruntime_providers_cuda.dll"); ASSERT_EQ(manifest->artifacts.size(), 3u); @@ -216,28 +216,28 @@ TEST(CudaEpManifestTest, WindowsArm64MetadataMatches20260806Bundle) { {"cudnn_ops64_9.dll", "c9e0ec0e0a4e659393e15897ed1f6e5bac677e0c0fe7e12290f0386f19477b6b"}, }); ExpectArtifact( - manifest->artifacts[2], "cuda-ep", "cuda-ep-bins-win-arm64-20260806-182803.zip", - "212e670c61b3292d4a7d98f16fc2cf61f7b080604e0c145e81c39ec81e7b3259", 96 * kMiB, + manifest->artifacts[2], "cuda-ep", "cuda-ep-bins-win-arm64-20260913-071350.zip", + "fc8f0a01daafedc82aa57072f023b11cf801dd94ed0a449ddc8ac4698e43e519", 96 * kMiB, { - {"onnxruntime-genai-cuda.dll", "ab61145f4bc6284286e663586f634b973072d58ced20c497c7e5259f2ef3fc08"}, - {"onnxruntime_providers_cuda.dll", "d92ffbd23a84f91b976baed9031de267efe1dc892d85c09d0979d25b89f5d1a0"}, + {"onnxruntime-genai-cuda.dll", "d27a02b8a83d0aaac47904d1baff4f7954aa25dcc130dba539fe9aafc53eab30"}, + {"onnxruntime_providers_cuda.dll", "82a3887c64791fc7131a0f705130f02f456952915ecea71372236656c59a47b2"}, }); ExpectUniqueInstalledPaths(*manifest); } -TEST(CudaEpManifestTest, LinuxX64MetadataMatches20260806Bundle) { +TEST(CudaEpManifestTest, LinuxX64MetadataMatches20260913Bundle) { const auto manifest = BuildCudaEpManifest(CudaEpPlatform::LinuxX64); ASSERT_TRUE(manifest.has_value()); - EXPECT_EQ(manifest->bundle_id, "cuda-ep-linux-x64-ort-1.28.0-genai-0.15.2-20260806-182830"); + EXPECT_EQ(manifest->bundle_id, "cuda-ep-linux-x64-ort-1.30.0-genai-0.16.0-20260913-071417"); EXPECT_EQ(manifest->provider_relative_path, "libonnxruntime_providers_cuda.so"); ASSERT_EQ(manifest->artifacts.size(), 1u); ExpectArtifact( - manifest->artifacts[0], "cuda-ep", "cuda-ep-linux-x64-20260806-182830.zip", - "abf347e7234d7434105efde12a2e0609fdd1d8828167b9873f4463926f1206e6", 448 * kMiB, + manifest->artifacts[0], "cuda-ep", "cuda-ep-linux-x64-20260913-071417.zip", + "482e615e4c66b3f14a980af00e330e28fa5897a34161ec2972f75289685546be", 448 * kMiB, { - {"libonnxruntime-genai-cuda.so", "d5300fc4413d9e74bd8dfceb5233fca6fcfa1d5ddc247081365fdb5f143091e6"}, - {"libonnxruntime_providers_cuda.so", "b88d7b7f4b2e81d3eff41663fc70f4ae9e03dee9e2301cb53dc250e5a96d7f7a"}, + {"libonnxruntime-genai-cuda.so", "3c95bb3b92f560718482dd3638388b6fd7374fb5ba6d9d5f4c4ca22c095e7681"}, + {"libonnxruntime_providers_cuda.so", "8589ae31ed4941e72693a6a2e9dd8b8e721c888cfd9e6ffcb26da4e6e58ce720"}, }); ExpectUniqueInstalledPaths(*manifest); } diff --git a/sdk_v2/deps_versions.json b/sdk_v2/deps_versions.json index db735f041..2559af4db 100644 --- a/sdk_v2/deps_versions.json +++ b/sdk_v2/deps_versions.json @@ -1,6 +1,6 @@ { "_comment": "Single source of truth for native dependency versions in sdk_v2. Read by sdk_v2/cpp/cmake/Find*.cmake and sdk_v2/python/_build_backend/__init__.py. The .pipelines/foundry-local-packaging.yml literals must match; the 'Validate pinned versions' step fails the build on drift.", - "onnxruntime": { "version": "1.28.0" }, - "onnxruntime-genai": { "version": "0.15.2" }, + "onnxruntime": { "version": "1.30.0" }, + "onnxruntime-genai": { "version": "0.16.0" }, "windows-ai-machinelearning": { "version": "2.1.70" } } diff --git a/sdk_v2/js/test/install-native.test.ts b/sdk_v2/js/test/install-native.test.ts index afc860d4d..aaef1f1a0 100644 --- a/sdk_v2/js/test/install-native.test.ts +++ b/sdk_v2/js/test/install-native.test.ts @@ -243,11 +243,11 @@ describe("downloadWithRetryAndRedirects", () => { describe("generateRestoreProjectXml", () => { it("includes bracketed exact versions for every artifact", () => { const xml = generateRestoreProjectXml([ - { name: "Microsoft.ML.OnnxRuntime", version: "1.28.0", expected: "onnxruntime.dll" }, - { name: "Microsoft.ML.OnnxRuntimeGenAI.Foundry", version: "0.15.2", expected: "onnxruntime-genai.dll" }, + { name: "Microsoft.ML.OnnxRuntime", version: "1.30.0", expected: "onnxruntime.dll" }, + { name: "Microsoft.ML.OnnxRuntimeGenAI.Foundry", version: "0.16.0", expected: "onnxruntime-genai.dll" }, ]); - expect(xml).toContain(''); - expect(xml).toContain(''); + expect(xml).toContain(''); + expect(xml).toContain(''); }); it("targets net8.0", () => { @@ -323,9 +323,9 @@ describe("findRestoredPackageDir", () => { }); it("finds the lowercased id/version directory", () => { - const dir = join(packagesDir, "microsoft.ml.onnxruntime", "1.28.0"); + const dir = join(packagesDir, "microsoft.ml.onnxruntime", "1.30.0"); mkdirSync(dir, { recursive: true }); - expect(findRestoredPackageDir(packagesDir, "Microsoft.ML.OnnxRuntime", "1.28.0")).toBe(dir); + expect(findRestoredPackageDir(packagesDir, "Microsoft.ML.OnnxRuntime", "1.30.0")).toBe(dir); }); it("throws when the expected package directory is missing", () => { @@ -341,13 +341,13 @@ describe("buildNugetInstallArgs", () => { it("passes exact package-only install flags and each configured source", () => { const args = buildNugetInstallArgs( { ...config, configFile: undefined }, - { id: "Microsoft.ML.OnnxRuntime", version: "1.28.0", outputDir: "pkgs" }, + { id: "Microsoft.ML.OnnxRuntime", version: "1.30.0", outputDir: "pkgs" }, ); expect(args).toEqual([ "install", "Microsoft.ML.OnnxRuntime", "-Version", - "1.28.0", + "1.30.0", "-OutputDirectory", "pkgs", "-NonInteractive", @@ -364,13 +364,13 @@ describe("buildNugetInstallArgs", () => { it("passes -ConfigFile and no -Source args when a config file is set", () => { const args = buildNugetInstallArgs( { ...config, configFile: "NuGet.config" }, - { id: "Microsoft.ML.OnnxRuntimeGenAI.Foundry", version: "0.15.2", outputDir: "pkgs" }, + { id: "Microsoft.ML.OnnxRuntimeGenAI.Foundry", version: "0.16.0", outputDir: "pkgs" }, ); expect(args).toEqual([ "install", "Microsoft.ML.OnnxRuntimeGenAI.Foundry", "-Version", - "0.15.2", + "0.16.0", "-OutputDirectory", "pkgs", "-NonInteractive", @@ -419,24 +419,24 @@ describe("findNugetPackageDir", () => { }); it("finds the id.version directory using nuget's original casing", () => { - const dir = join(outputDir, "Microsoft.ML.OnnxRuntime.1.28.0"); + const dir = join(outputDir, "Microsoft.ML.OnnxRuntime.1.30.0"); mkdirSync(dir, { recursive: true }); - expect(findNugetPackageDir(outputDir, "Microsoft.ML.OnnxRuntime", "1.28.0")).toBe(dir); + expect(findNugetPackageDir(outputDir, "Microsoft.ML.OnnxRuntime", "1.30.0")).toBe(dir); }); it("matches case-insensitively regardless of the casing nuget.exe produced", () => { - const dir = join(outputDir, "microsoft.ml.onnxruntimegenai.foundry.0.15.2"); + const dir = join(outputDir, "microsoft.ml.onnxruntimegenai.foundry.0.16.0"); mkdirSync(dir, { recursive: true }); - expect(findNugetPackageDir(outputDir, "Microsoft.ML.OnnxRuntimeGenAI.Foundry", "0.15.2")).toBe(dir); + expect(findNugetPackageDir(outputDir, "Microsoft.ML.OnnxRuntimeGenAI.Foundry", "0.16.0")).toBe(dir); }); it("only looks at immediate children of outputDir, not nested dependency package folders", () => { // Simulates nuget.exe restoring a transitive dependency alongside the requested package — // only the exact id.version match at the root should be returned. mkdirSync(join(outputDir, "Some.Other.Dependency.2.0.0"), { recursive: true }); - const dir = join(outputDir, "Microsoft.ML.OnnxRuntime.1.28.0"); + const dir = join(outputDir, "Microsoft.ML.OnnxRuntime.1.30.0"); mkdirSync(join(dir, "runtimes", "win-x64", "native"), { recursive: true }); - expect(findNugetPackageDir(outputDir, "Microsoft.ML.OnnxRuntime", "1.28.0")).toBe(dir); + expect(findNugetPackageDir(outputDir, "Microsoft.ML.OnnxRuntime", "1.30.0")).toBe(dir); }); it("throws when the expected package directory is missing", () => { diff --git a/sdk_v2/rust/deps_versions.json b/sdk_v2/rust/deps_versions.json index ccc2af55e..7030dd39f 100644 --- a/sdk_v2/rust/deps_versions.json +++ b/sdk_v2/rust/deps_versions.json @@ -1,6 +1,6 @@ { "_comment": "Synced copy of sdk_v2/deps_versions.json, committed here so the pinned native versions ship inside the published crate (Cargo cannot include files outside the package root). The canonical file at sdk_v2/deps_versions.json remains the single source of truth; build.rs fails the build if the version pins drift. When updating versions, edit the canonical file and re-copy it here.", - "onnxruntime": { "version": "1.28.0" }, - "onnxruntime-genai": { "version": "0.15.2" }, + "onnxruntime": { "version": "1.30.0" }, + "onnxruntime-genai": { "version": "0.16.0" }, "windows-ai-machinelearning": { "version": "2.1.70" } } From fa36bd38edd4eb83bc4a3c418094af66d0c99196 Mon Sep 17 00:00:00 2001 From: Baiju Meswani Date: Sun, 13 Sep 2026 17:12:10 +0000 Subject: [PATCH 2/5] fix test --- sdk_v2/cpp/test/sdk_api/chat_session_test.cc | 51 +++++++------------- 1 file changed, 17 insertions(+), 34 deletions(-) diff --git a/sdk_v2/cpp/test/sdk_api/chat_session_test.cc b/sdk_v2/cpp/test/sdk_api/chat_session_test.cc index 7329c22f6..7ad900aa0 100644 --- a/sdk_v2/cpp/test/sdk_api/chat_session_test.cc +++ b/sdk_v2/cpp/test/sdk_api/chat_session_test.cc @@ -819,10 +819,8 @@ TEST_F(ModelFixture, OpenAIJsonMultipleRequestsAreStateless) { } // ------------------------------------------------------------------------ -// JSON request with a tools[] array. Model behavior varies by version — -// it may invoke the tool or just answer in prose. Both shapes are valid; -// we assert only that the response parses and contains *some* assistant -// signal (tool_calls OR content). +// JSON request with a tools[] array and required tool choice. Requiring a call keeps this API contract test +// deterministic across model and runtime versions while exercising JSON conversion, guidance, and response parsing. // ------------------------------------------------------------------------ TEST_F(ToolCallFixture, OpenAIJsonWithToolDefinition) { using namespace foundry_local; @@ -850,6 +848,7 @@ TEST_F(ToolCallFixture, OpenAIJsonWithToolDefinition) { {"second", {{"type", "integer"}, {"description", "second number"}}}}}, {"required", json::array({"first", "second"})}}}}}}, })}, + {"tool_choice", "required"}, {"temperature", 0}, {"max_tokens", 256}, }; @@ -863,41 +862,25 @@ TEST_F(ToolCallFixture, OpenAIJsonWithToolDefinition) { ASSERT_EQ(chat_response.choices.size(), 1u); const auto& msg = chat_response.choices[0].message; - bool has_tool_calls = msg.tool_calls.has_value() && !msg.tool_calls->empty(); - bool has_content = msg.content.has_value() && !msg.content->empty(); + ASSERT_TRUE(msg.tool_calls.has_value()); + ASSERT_FALSE(msg.tool_calls->empty()); - ASSERT_TRUE(has_tool_calls || has_content) - << "Expected either tool_calls or content in assistant response"; + // Validate the call shape — correct function name and the two integer arguments {7, 6}. Argument order is + // intentionally flexible because multiplication is commutative and models may legitimately swap the operands. + const auto& call = (*msg.tool_calls)[0]; + EXPECT_EQ(call.function.name, "multiply_numbers") << "Model invoked unexpected tool: " << call.function.name; - if (has_tool_calls) { - // Tool-calling path: the model invoked our tool. Validate the call shape - // matches the contract — correct function name and the two integer - // arguments {7, 6} (order-insensitive: model may legitimately swap). - const auto& call = (*msg.tool_calls)[0]; - EXPECT_EQ(call.function.name, "multiply_numbers") - << "Model invoked unexpected tool: " << call.function.name; + auto args = nlohmann::json::parse(call.function.arguments); + ASSERT_TRUE(args.contains("first")) << "Tool args missing 'first': " << call.function.arguments; + ASSERT_TRUE(args.contains("second")) << "Tool args missing 'second': " << call.function.arguments; - auto args = nlohmann::json::parse(call.function.arguments); - ASSERT_TRUE(args.contains("first")) << "Tool args missing 'first': " << call.function.arguments; - ASSERT_TRUE(args.contains("second")) << "Tool args missing 'second': " << call.function.arguments; + int first = args["first"].get(); + int second = args["second"].get(); - int first = args["first"].get(); - int second = args["second"].get(); + EXPECT_TRUE((first == 7 && second == 6) || (first == 6 && second == 7)) + << "Tool args don't match the prompt (7 * 6). Got first=" << first << ", second=" << second; - EXPECT_TRUE((first == 7 && second == 6) || (first == 6 && second == 7)) - << "Tool args don't match the prompt (7 * 6). Got first=" << first - << ", second=" << second; - - std::cout << "Tool-call JSON path: model invoked multiply_numbers(" << first << ", " << second << ")\n"; - } else { - // Direct-answer path: model declined to call the tool. Reply must still - // contain the correct product, otherwise the model didn't actually answer - // the question. - EXPECT_NE(msg.content->find("42"), std::string::npos) - << "Direct-answer reply missing '42'. Got: " << *msg.content; - - std::cout << "Tool-call JSON path: model produced content: " << *msg.content << "\n"; - } + std::cout << "Tool-call JSON path: model invoked multiply_numbers(" << first << ", " << second << ")\n"; } // ------------------------------------------------------------------------ From 281ea0d1cc62b62478af4642e751bca57e493f7e Mon Sep 17 00:00:00 2001 From: Baiju Meswani Date: Sun, 13 Sep 2026 18:06:19 +0000 Subject: [PATCH 3/5] fix test --- .../tests/integration/chat_client_test.rs | 88 +++++++++++++------ 1 file changed, 59 insertions(+), 29 deletions(-) diff --git a/sdk_v2/rust/tests/integration/chat_client_test.rs b/sdk_v2/rust/tests/integration/chat_client_test.rs index b27f66b4e..dcb36f53f 100644 --- a/sdk_v2/rust/tests/integration/chat_client_test.rs +++ b/sdk_v2/rust/tests/integration/chat_client_test.rs @@ -6,6 +6,7 @@ use foundry_local_sdk::{ ChatCompletionRequestUserMessage, ChatToolChoice, }; use serde_json::json; +use std::collections::BTreeMap; use std::sync::Arc; use tokio_stream::StreamExt; @@ -242,15 +243,20 @@ async fn should_perform_tool_calling_chat_completion_streaming() { let (client, model) = setup_chat_client().await; let client = client.tool_choice(ChatToolChoice::Required); + #[derive(Default)] + struct StreamedToolCall { + id: String, + name: String, + arguments: String, + } + let tools = vec![common::get_multiply_tool()]; let mut messages = vec![ system_message("You are a math assistant. Use the multiply tool to answer."), user_message("What is 6 times 7?"), ]; - let mut tool_call_name = String::new(); - let mut tool_call_args = String::new(); - let mut tool_call_id = String::new(); + let mut streamed_tool_calls = BTreeMap::::new(); let mut stream = client .complete_streaming_chat(&messages, Some(&tools)) @@ -262,52 +268,76 @@ async fn should_perform_tool_calling_chat_completion_streaming() { if let Some(choice) = chunk.choices.first() { if let Some(ref tool_calls) = choice.delta.tool_calls { for call in tool_calls { + let accumulated = streamed_tool_calls.entry(call.index).or_default(); if let Some(ref func) = call.function { if let Some(ref name) = func.name { - tool_call_name.push_str(name); + accumulated.name.push_str(name); } if let Some(ref args) = func.arguments { - tool_call_args.push_str(args); + accumulated.arguments.push_str(args); } } if let Some(ref id) = call.id { - tool_call_id = id.clone(); + accumulated.id.push_str(id); } } } } } - assert_eq!( - tool_call_name, "multiply", - "Expected streamed tool call to 'multiply'" + + assert!( + !streamed_tool_calls.is_empty(), + "Expected at least one streamed tool call" ); - let args: serde_json::Value = - serde_json::from_str(&tool_call_args).unwrap_or_else(|_| json!({})); - let a = args["a"].as_f64().unwrap_or(0.0); - let b = args["b"].as_f64().unwrap_or(0.0); - let product = (a * b) as i64; + for call in streamed_tool_calls.values() { + assert_eq!( + call.name, "multiply", + "Expected streamed tool call to 'multiply'" + ); + let args: serde_json::Value = serde_json::from_str(&call.arguments) + .expect("streamed tool arguments should be valid JSON"); + let a = args["a"] + .as_f64() + .expect("multiply argument 'a' should be numeric"); + let b = args["b"] + .as_f64() + .expect("multiply argument 'b' should be numeric"); + assert_eq!( + (a * b) as i64, + 42, + "streamed multiply arguments should produce 42" + ); + } + let assistant_tool_calls = streamed_tool_calls + .values() + .map(|call| { + json!({ + "id": call.id, + "type": "function", + "function": { + "name": call.name, + "arguments": call.arguments + } + }) + }) + .collect::>(); let assistant_msg: ChatCompletionRequestMessage = serde_json::from_value(json!({ "role": "assistant", - "tool_calls": [{ - "id": tool_call_id, - "type": "function", - "function": { - "name": tool_call_name, - "arguments": tool_call_args - } - }] + "tool_calls": assistant_tool_calls })) .expect("failed to construct assistant message"); messages.push(assistant_msg); - messages.push( - ChatCompletionRequestToolMessage { - content: product.to_string().into(), - tool_call_id: tool_call_id.clone(), - } - .into(), - ); + for call in streamed_tool_calls.values() { + messages.push( + ChatCompletionRequestToolMessage { + content: "42".into(), + tool_call_id: call.id.clone(), + } + .into(), + ); + } messages.push(system_message( "Respond only with the answer generated by the tool. Do not call any tools.", )); From 6463d1bd6188e6c6881dbfec4e7c8d5be330a40f Mon Sep 17 00:00:00 2001 From: Baiju Meswani Date: Sun, 13 Sep 2026 20:09:01 +0000 Subject: [PATCH 4/5] C# SDK tests --- .../ChatCompletionsTests.cs | 121 +++++++++--------- 1 file changed, 62 insertions(+), 59 deletions(-) diff --git a/sdk_v2/cs/test/FoundryLocal.Tests/ChatCompletionsTests.cs b/sdk_v2/cs/test/FoundryLocal.Tests/ChatCompletionsTests.cs index 0bd8c7c24..66afb5da8 100644 --- a/sdk_v2/cs/test/FoundryLocal.Tests/ChatCompletionsTests.cs +++ b/sdk_v2/cs/test/FoundryLocal.Tests/ChatCompletionsTests.cs @@ -236,39 +236,39 @@ public async Task DirectTool_NoStreaming_Succeeds() await Assert.That(response.Choices[0].Message).IsNotNull(); await Assert.That(response.Choices[0].Message.ToolCalls).IsNotNull().And.IsNotEmpty(); - await Assert.That(response.Choices[0].Message.ToolCalls?.Count).IsEqualTo(1); - await Assert.That(response.Choices[0].Message.ToolCalls?[0].Type).IsEqualTo("function"); - await Assert.That(response.Choices[0].Message.ToolCalls?[0].FunctionCall?.Name).IsEqualTo("multiply_numbers"); - - var expected = new Dictionary + var assistantToolCalls = response.Choices[0].Message.ToolCalls!; + foreach (var assistantToolCall in assistantToolCalls) { - ["first"] = 7, - ["second"] = 6 - }; - - var json = response.Choices[0].Message.ToolCalls?[0].FunctionCall?.Arguments; - var actual = System.Text.Json.JsonSerializer.Deserialize>(json!); - await Assert.That(actual).IsEquivalentTo(expected); + await Assert.That(assistantToolCall.Type).IsEqualTo("function"); + await Assert.That(assistantToolCall.FunctionCall?.Name).IsEqualTo("multiply_numbers"); + await Assert.That(assistantToolCall.Id).IsNotNull(); + await Assert.That(assistantToolCall.Id).IsNotEmpty(); + + var json = assistantToolCall.FunctionCall?.Arguments; + var actual = System.Text.Json.JsonSerializer.Deserialize>(json!); + await Assert.That(actual).IsNotNull(); + await Assert.That(actual!.Keys).Contains("first"); + await Assert.That(actual.Keys).Contains("second"); + await Assert.That(actual["first"] * actual["second"]).IsEqualTo(42); + } - // Replay the assistant turn that issued the call before answering it. A Chat Completions + // Replay the assistant turn that issued the calls before answering them. A Chat Completions // payload is self-contained — every request is correlated against an empty transcript — so a - // tool result is only matched to its call when that call travels with it. Content stays null + // tool result is only matched to its call when every call travels with it. Content stays null // so the replay matches the content-free turn the model generated, rather than feeding its // tool-call marker text back. - var assistantToolCall = response.Choices[0].Message.ToolCalls![0]; - await Assert.That(assistantToolCall.Id).IsNotNull(); - await Assert.That(assistantToolCall.Id).IsNotEmpty(); - - messages.Add(new ChatMessage { Role = "assistant", ToolCalls = [assistantToolCall] }); + messages.Add(new ChatMessage { Role = "assistant", ToolCalls = assistantToolCalls }); - // Add the response from invoking the tool call to the conversation and check if the model can continue correctly - var toolCallResponse = "7 x 6 = 42."; - messages.Add(new ChatMessage + // Add one correlated result for every invocation and check if the model can continue correctly. + foreach (var assistantToolCall in assistantToolCalls) { - Role = "tool", - ToolCallId = assistantToolCall.Id, - Content = toolCallResponse - }); + messages.Add(new ChatMessage + { + Role = "tool", + ToolCallId = assistantToolCall.Id, + Content = "7 x 6 = 42." + }); + } // Prompt the model to continue the conversation after the tool call messages.Add(new ChatMessage { Role = "system", Content = "Respond only with the answer generated by the tool." }); @@ -331,7 +331,7 @@ public async Task DirectTool_Streaming_Succeeds() var isFirstChunk = true; bool gotFinishReason = false; StringBuilder responseMessage = new(); - ChatCompletionCreateResponse? toolCallResponse = null; + List toolCallResponses = []; var validateResponse = async (ChatCompletionCreateResponse? response) => { @@ -355,9 +355,10 @@ public async Task DirectTool_Streaming_Succeeds() } else { - // we only expect one callback with all the args so we're not accumulating across multiple delta chunks - await Assert.That(toolCallResponse).IsNull(); - toolCallResponse = response; + if (delta.ToolCalls is { Count: > 0 }) + { + toolCallResponses.Add(response); + } } }; @@ -368,39 +369,41 @@ public async Task DirectTool_Streaming_Succeeds() } await Assert.That(gotFinishReason).IsTrue(); - await Assert.That(toolCallResponse).IsNotNull(); - await Assert.That(toolCallResponse!.Choices.Count).IsEqualTo(1); - await Assert.That(toolCallResponse.Choices[0].Delta.ToolCalls).IsNotNull(); - await Assert.That(toolCallResponse.Choices[0].Delta.ToolCalls?.Count).IsEqualTo(1); - await Assert.That(toolCallResponse.Choices[0].Delta.ToolCalls?[0].Type).IsEqualTo("function"); - await Assert.That(toolCallResponse.Choices[0].Delta.ToolCalls?[0].FunctionCall?.Name).IsEqualTo("multiply_numbers"); - - var expected = new Dictionary - { - ["first"] = 7, - ["second"] = 6 - }; + await Assert.That(toolCallResponses).IsNotEmpty(); + var streamedToolCalls = toolCallResponses + .SelectMany(response => response.Choices[0].Delta.ToolCalls!) + .ToList(); + await Assert.That(streamedToolCalls).IsNotEmpty(); - var json = toolCallResponse.Choices[0].Message.ToolCalls?[0].FunctionCall?.Arguments; - var actual = System.Text.Json.JsonSerializer.Deserialize>(json!); - await Assert.That(actual).IsEquivalentTo(expected); - - // Replay the assistant turn that issued the call before answering it — see - // DirectTool_NoStreaming_Succeeds for why the call has to travel with its result. - var streamedToolCall = toolCallResponse.Choices[0].Delta.ToolCalls![0]; - await Assert.That(streamedToolCall.Id).IsNotNull(); - await Assert.That(streamedToolCall.Id).IsNotEmpty(); + foreach (var streamedToolCall in streamedToolCalls) + { + await Assert.That(streamedToolCall.Type).IsEqualTo("function"); + await Assert.That(streamedToolCall.FunctionCall?.Name).IsEqualTo("multiply_numbers"); + await Assert.That(streamedToolCall.Id).IsNotNull(); + await Assert.That(streamedToolCall.Id).IsNotEmpty(); + + var json = streamedToolCall.FunctionCall?.Arguments; + var actual = System.Text.Json.JsonSerializer.Deserialize>(json!); + await Assert.That(actual).IsNotNull(); + await Assert.That(actual!.Keys).Contains("first"); + await Assert.That(actual.Keys).Contains("second"); + await Assert.That(actual["first"] * actual["second"]).IsEqualTo(42); + } - messages.Add(new ChatMessage { Role = "assistant", ToolCalls = [streamedToolCall] }); + // Replay the assistant turn that issued the calls before answering them — see + // DirectTool_NoStreaming_Succeeds for why every call has to travel with its result. + messages.Add(new ChatMessage { Role = "assistant", ToolCalls = streamedToolCalls }); - // Add the response from invoking the tool call to the conversation and check if the model can continue correctly - var toolResponse = "7 x 6 = 42."; - messages.Add(new ChatMessage + // Add one correlated result for every invocation and check if the model can continue correctly. + foreach (var streamedToolCall in streamedToolCalls) { - Role = "tool", - ToolCallId = streamedToolCall.Id, - Content = toolResponse - }); + messages.Add(new ChatMessage + { + Role = "tool", + ToolCallId = streamedToolCall.Id, + Content = "7 x 6 = 42." + }); + } // Prompt the model to continue the conversation after the tool call messages.Add(new ChatMessage { Role = "system", Content = "Respond only with the answer generated by the tool." }); From 57d176094bcfcf007f0d7bc1c30ad63e76649532 Mon Sep 17 00:00:00 2001 From: Baiju Meswani Date: Mon, 14 Sep 2026 16:38:02 +0000 Subject: [PATCH 5/5] update macos test ort dylib version --- sdk_v2/python/test/unit/test_lib_loader.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/sdk_v2/python/test/unit/test_lib_loader.py b/sdk_v2/python/test/unit/test_lib_loader.py index 5b14eaf87..d5508ce1b 100644 --- a/sdk_v2/python/test/unit/test_lib_loader.py +++ b/sdk_v2/python/test/unit/test_lib_loader.py @@ -136,7 +136,7 @@ def test_find_file_in_package_supports_vanilla_windows_capi_layout(self, tmp_pat def test_find_file_in_package_supports_versioned_macos_ort_dylib(self, tmp_path, monkeypatch): pkg_spec = self._install_fake_package(tmp_path, "onnxruntime") - dylib_path = tmp_path / "onnxruntime" / "capi" / "libonnxruntime.1.28.0.dylib" + dylib_path = tmp_path / "onnxruntime" / "capi" / "libonnxruntime.1.30.0.dylib" dylib_path.parent.mkdir() dylib_path.write_text("", encoding="utf-8")