chore: import upstream snapshot with attribution

This commit is contained in:
wehub-resource-sync
2026-07-13 13:18:33 +08:00
commit 4ececc111a
2017 changed files with 331736 additions and 0 deletions
+149
View File
@@ -0,0 +1,149 @@
// Copyright (c) Microsoft Corporation.
// SPDX-License-Identifier: Apache-2.0
// DeepSpeed Team
#include <cassert>
#include "memory_access_utils.h"
#include "spatial_cuda_layers.h"
/*
Fused bias add variants
*/
namespace badd_opt {
constexpr int threads = 256;
constexpr int steps = 2;
constexpr int granularity = 16;
constexpr int vals_per_h = granularity / sizeof(__half);
constexpr int vals_per_h2 = granularity / sizeof(__half2);
constexpr int vals_per_block = threads * steps * vals_per_h;
constexpr int stride = vals_per_h * threads;
} // namespace badd_opt
__global__ void opt_bias_add(__half* result,
const __half* activation,
const __half* bias,
int seq_len,
int channels)
{
const int id = blockIdx.x * badd_opt::vals_per_block + threadIdx.x * badd_opt::vals_per_h;
const int stride = badd_opt::vals_per_h * badd_opt::threads;
for (int i = 0; i < badd_opt::steps; i++) {
if (id + i * badd_opt::stride < seq_len * channels) {
__half2 act_buffer[badd_opt::vals_per_h2];
__half2 bias_buffer[badd_opt::vals_per_h2];
mem_access::load_global<badd_opt::granularity>(act_buffer,
activation + id + i * stride);
mem_access::load_global<badd_opt::granularity>(bias_buffer,
bias + ((id + i * stride) % channels));
for (int j = 0; j < badd_opt::vals_per_h2; j++) { act_buffer[j] += bias_buffer[j]; }
mem_access::store_global<badd_opt::granularity>(result + id + i * stride, act_buffer);
}
}
}
__global__ void opt_bias_add_add(__half* result,
const __half* activation,
const __half* bias,
const __half* other,
int seq_len,
int channels)
{
const int id = blockIdx.x * badd_opt::vals_per_block + threadIdx.x * badd_opt::vals_per_h;
const int stride = badd_opt::vals_per_h * badd_opt::threads;
for (int i = 0; i < badd_opt::steps; i++) {
if (id + i * badd_opt::stride < seq_len * channels) {
__half2 act_buffer[badd_opt::vals_per_h2];
__half2 bias_buffer[badd_opt::vals_per_h2];
__half2 other_buffer[badd_opt::vals_per_h2];
mem_access::load_global<badd_opt::granularity>(act_buffer,
activation + id + i * stride);
mem_access::load_global<badd_opt::granularity>(bias_buffer,
bias + ((id + i * stride) % channels));
mem_access::load_global<badd_opt::granularity>(other_buffer, other + id + i * stride);
for (int j = 0; j < badd_opt::vals_per_h2; j++) {
act_buffer[j] += bias_buffer[j] + other_buffer[j];
}
mem_access::store_global<badd_opt::granularity>(result + id + i * stride, act_buffer);
}
}
}
__global__ void opt_bias_add_bias_add(__half* result,
const __half* activation,
const __half* bias,
const __half* other,
const __half* other_bias,
int seq_len,
int channels)
{
const int id = blockIdx.x * badd_opt::vals_per_block + threadIdx.x * badd_opt::vals_per_h;
const int stride = badd_opt::vals_per_h * badd_opt::threads;
for (int i = 0; i < badd_opt::steps; i++) {
if (id + i * badd_opt::stride < seq_len * channels) {
__half2 act_buffer[badd_opt::vals_per_h2];
__half2 bias_buffer[badd_opt::vals_per_h2];
__half2 other_buffer[badd_opt::vals_per_h2];
__half2 other_bias_buffer[badd_opt::vals_per_h2];
mem_access::load_global<badd_opt::granularity>(act_buffer,
activation + id + i * stride);
mem_access::load_global<badd_opt::granularity>(bias_buffer,
bias + ((id + i * stride) % channels));
mem_access::load_global<badd_opt::granularity>(other_buffer, other + id + i * stride);
mem_access::load_global<badd_opt::granularity>(
other_bias_buffer, other_bias + ((id + i * stride) % channels));
for (int j = 0; j < badd_opt::vals_per_h2; j++) {
act_buffer[j] =
(act_buffer[j] + bias_buffer[j]) + (other_buffer[j] + other_bias_buffer[j]);
}
mem_access::store_global<badd_opt::granularity>(result + id + i * stride, act_buffer);
}
}
}
void launch_opt_bias_add(__half* result,
const __half* activation,
const __half* bias,
const __half* other,
const __half* other_bias,
int batch_size,
int seq_len,
int channels,
cudaStream_t stream)
{
// Should evaluate `true` for reasonable hidden sizes
assert(channels % badd_opt::vals_per_h == 0);
const int effective_seq_len = batch_size * seq_len;
const int vals = effective_seq_len * channels;
dim3 block(badd_opt::threads);
dim3 grid((vals + badd_opt::vals_per_block - 1) / badd_opt::vals_per_block);
if (!other) {
// We shouldn't have a bias if there's no activation
assert(!other_bias);
opt_bias_add<<<grid, block, 0, stream>>>(
result, activation, bias, effective_seq_len, channels);
} else if (!other_bias) {
opt_bias_add_add<<<grid, block, 0, stream>>>(
result, activation, bias, other, effective_seq_len, channels);
} else {
opt_bias_add_bias_add<<<grid, block, 0, stream>>>(
result, activation, bias, other, other_bias, effective_seq_len, channels);
}
}
+112
View File
@@ -0,0 +1,112 @@
// Copyright (c) Microsoft Corporation.
// SPDX-License-Identifier: Apache-2.0
// DeepSpeed Team
#include <c10/cuda/CUDAStream.h>
#include <torch/extension.h>
#include <cstdio>
#include <vector>
#include "spatial_cuda_layers.h"
ChannelsLastProblem dimension_problem(at::Tensor& input)
{
ChannelsLastProblem dims;
if (input.dim() == 4) {
// In some sense this is unsafe (and a reflection of the assumptions made inside
// the C10 options checker). Basically, there's no great way to be sure that
// a tensor is in channels last because a 1x1 image will appear to be in channels
// last even when it isn't.
assert(input.is_contiguous(at::MemoryFormat::ChannelsLast));
dims.batch_size = input.size(0);
dims.seq_len = input.size(2) * input.size(3);
dims.channels = input.size(1);
} else {
assert(input.is_contiguous());
dims.batch_size = input.size(0);
dims.seq_len = input.size(1);
dims.channels = input.size(2);
}
return dims;
}
at::Tensor seq_unroll_bias_add(at::Tensor& input, at::Tensor& bias)
{
assert(input.dtype() == at::kHalf);
// TODO(cmikeh2): Should probably refactor this into a more portable
// description, since it does generalize for channels-last
ChannelsLastProblem problem = dimension_problem(input);
auto output = at::empty_like(input);
launch_opt_bias_add((__half*)output.data_ptr(),
(const __half*)input.data_ptr(),
(const __half*)bias.data_ptr(),
nullptr,
nullptr,
problem.batch_size,
problem.seq_len,
problem.channels,
at::cuda::getCurrentCUDAStream());
return output;
}
at::Tensor seq_bias_add_add(at::Tensor& input, at::Tensor& bias, at::Tensor& other)
{
assert(input.dtype() == at::kHalf);
// TODO(cmikeh2): Should probably refactor this into a more portable
// description, since it does generalize for channels-last
ChannelsLastProblem problem = dimension_problem(input);
auto output = at::empty_like(input);
launch_opt_bias_add((__half*)output.data_ptr(),
(const __half*)input.data_ptr(),
(const __half*)bias.data_ptr(),
(const __half*)other.data_ptr(),
nullptr,
problem.batch_size,
problem.seq_len,
problem.channels,
at::cuda::getCurrentCUDAStream());
return output;
}
at::Tensor seq_bias_add_bias_add(at::Tensor& input,
at::Tensor& bias,
at::Tensor& other,
at::Tensor& other_bias)
{
assert(input.dtype() == at::kHalf);
// TODO(cmikeh2): Should probably refactor this into a more portable
// description, since it does generalize for channels-last
ChannelsLastProblem problem = dimension_problem(input);
auto output = at::empty_like(input);
launch_opt_bias_add((__half*)output.data_ptr(),
(const __half*)input.data_ptr(),
(const __half*)bias.data_ptr(),
(const __half*)other.data_ptr(),
(const __half*)other_bias.data_ptr(),
problem.batch_size,
problem.seq_len,
problem.channels,
at::cuda::getCurrentCUDAStream());
return output;
}
PYBIND11_MODULE(TORCH_EXTENSION_NAME, m)
{
m.def("nhwc_bias_add", &seq_unroll_bias_add);
m.def("nhwc_bias_add_add", &seq_bias_add_add);
m.def("nhwc_bias_add_bias_add", &seq_bias_add_bias_add);
}
@@ -0,0 +1,37 @@
// Copyright (c) Microsoft Corporation.
// SPDX-License-Identifier: Apache-2.0
// DeepSpeed Team
#pragma once
#if __CUDA_ARCH__ >= 530
#define HALF_PRECISION_AVAILABLE = 1
#endif
#ifdef __HIP_PLATFORM_AMD__
#include <hip/hip_cooperative_groups.h>
#else
#include <cooperative_groups.h>
#endif
#include <cuda.h>
#include <cuda_fp16.h>
/*********** Group Norm Kernels, Structs, and Helpers ************/
struct {
int64_t batch_size;
int64_t seq_len;
int64_t channels;
} typedef ChannelsLastProblem;
void launch_opt_bias_add(__half* result,
const __half* activation,
const __half* bias,
const __half* other,
const __half* other_bias,
int batch_size,
int seq_len,
int channels,
cudaStream_t stream);