chore: import upstream snapshot with attribution
CodeQL / Analyze (csharp) (push) Has been cancelled
CodeQL / Analyze (python) (push) Has been cancelled

This commit is contained in:
wehub-resource-sync
2026-07-13 13:21:23 +08:00
commit b957a53def
5423 changed files with 863745 additions and 0 deletions
@@ -0,0 +1,20 @@
<Project Sdk="Microsoft.NET.Sdk">
<PropertyGroup>
<OutputType>Exe</OutputType>
<TargetFramework>net10.0</TargetFramework>
<NoWarn>$(NoWarn);CA2007,CA2208,CS1591,CA1024,IDE0009,IDE0055,IDE0073,IDE0211,VSTHRD111,SKEXP0001</NoWarn>
</PropertyGroup>
<ItemGroup>
<!--
TODO: fix this WORKAROUND
CUDA provider set up with Microsoft.ML.OnnxRuntimeGenAI.Cuda 0.8.3 + Microsoft.ML.OnnxRuntime.Gpu 1.22.1
- doesn't work with Microsoft.ML.OnnxRuntime 1.22.1
- works with Microsoft.ML.OnnxRuntime 1.22.0
-->
<PackageReference Include="Microsoft.ML.OnnxRuntime" VersionOverride="1.22.0" NoWarn="NU1605"/>
<PackageReference Include="Microsoft.ML.OnnxRuntime.Gpu" />
<PackageReference Include="Microsoft.ML.OnnxRuntimeGenAI.Cuda"/>
<ProjectReference Include="..\..\..\src\Connectors\Connectors.Onnx\Connectors.Onnx.csproj"/>
<ProjectReference Include="..\..\..\src\SemanticKernel.Abstractions\SemanticKernel.Abstractions.csproj"/>
</ItemGroup>
</Project>
@@ -0,0 +1,48 @@
using System;
using System.Collections.Generic;
using Microsoft.Extensions.AI;
using Microsoft.SemanticKernel;
using Microsoft.SemanticKernel.Connectors.Onnx;
// Path to the folder of your downloaded ONNX CUDA model
// i.e: D:\repo\huggingface\Phi-3-mini-4k-instruct-onnx\cuda\cuda-int4-rtn-block-32
string modelPath = "MODEL_PATH";
IKernelBuilder builder = Kernel.CreateBuilder();
builder.AddOnnxRuntimeGenAIChatClient(
modelPath: modelPath,
// Specify the provider you want to use, e.g., "cuda" for GPU support
// For other execution providers, check: https://onnxruntime.ai/docs/genai/reference/config#provideroptions
providers: [new Provider("cuda")] //
);
Kernel kernel = builder.Build();
using IChatClient chatClient = kernel.GetRequiredService<IChatClient>();
List<ChatMessage> chatHistory = [];
while (true)
{
Console.Write("User > ");
string userMessage = Console.ReadLine()!;
if (string.IsNullOrEmpty(userMessage))
{
break;
}
chatHistory.Add(new ChatMessage(ChatRole.User, userMessage));
try
{
ChatResponse result = await chatClient.GetResponseAsync(chatHistory, new() { MaxOutputTokens = 1024 });
Console.WriteLine($"Assistant > {result.Text}");
chatHistory.AddRange(result.Messages);
}
catch (Exception e)
{
Console.WriteLine(e.Message);
}
}
@@ -0,0 +1,44 @@
# Onnx Simple Chat with Cuda Execution Provider
This sample demonstrates how you use ONNX Connector with CUDA Execution Provider to run Local Models straight from files using Semantic Kernel.
In this example we setup Chat Client from ONNX Connector with [Microsoft's Phi-3-ONNX](https://huggingface.co/microsoft/Phi-3-mini-4k-instruct-onnx) model
> [!IMPORTANT]
> You can modify to use any other combination of models enabled for ONNX runtime.
## Semantic Kernel used Features
- [Chat Client](https://github.com/microsoft/semantic-kernel/blob/main/dotnet/src/SemanticKernel.Abstractions/AI/ChatCompletion/IChatCompletionService.cs) - Using the Chat Completion Service from [Onnx Connector](https://github.com/microsoft/semantic-kernel/blob/main/dotnet/src/Connectors/Connectors.Onnx/OnnxRuntimeGenAIChatCompletionService.cs) to generate responses from the Local Model.
## Prerequisites
- [.NET 10](https://dotnet.microsoft.com/download/dotnet/10.0).
- [NVIDIA GPU](https://www.nvidia.com/en-us/geforce/graphics-cards)
- [NVIDIA CUDA v12 Toolkit](https://developer.nvidia.com/cuda-12-0-0-download-archive)
- [NVIDIA cuDNN v9.11](https://developer.nvidia.com/cudnn-9-11-0-download-archive)
- Windows users only:
Ensure `PATH` environment variable includes the `bin` folder of the CUDA Toolkit and cuDNN.
i.e:
- C:\Program Files\NVIDIA GPU Computing Toolkit\CUDA\v12.0\bin
- C:\Program Files\NVIDIA\CUDNN\v9.11\bin\12.9
- Downloaded ONNX Models (see below).
## Downloading the Model
For this example we chose Hugging Face as our repository for download of the local models, go to a directory of your choice where the models should be downloaded and run the following commands:
```powershell
git lfs install
git clone https://huggingface.co/microsoft/Phi-3-mini-4k-instruct-onnx
```
Update the `Program.cs` file lines below with the paths to the models you downloaded in the previous step.
```csharp
// i.e. Running on Windows
string modelPath = "D:\\repo\\huggingface\\Phi-3-mini-4k-instruct-onnx\\cuda\\cuda-int4-rtn-block-32";
```