// Copyright (c) 2023 PaddlePaddle Authors. All Rights Reserved. // // Licensed under the Apache License, Version 2.0 (the "License"); // you may not use this file except in compliance with the License. // You may obtain a copy of the License at // // http://www.apache.org/licenses/LICENSE-2.0 // // Unless required by applicable law or agreed to in writing, software // distributed under the License is distributed on an "AS IS" BASIS, // WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. // See the License for the specific language governing permissions and // limitations under the License. #include "paddle/fluid/framework/ir/xpu/quant_utils.h" #include #include "paddle/fluid/framework/ir/quantize_helper.h" #include "paddle/phi/api/lib/data_transform.h" #include "paddle/phi/backends/xpu/xpu_info.h" #include "paddle/phi/core/enforce.h" #include "paddle/phi/core/platform/device_context.h" #include "paddle/phi/kernels/assign_kernel.h" #include "paddle/phi/kernels/cast_kernel.h" #include "paddle/phi/kernels/transpose_kernel.h" namespace paddle { namespace framework { namespace ir { void Assign(const DenseTensor& in, DenseTensor* out) { auto* cpu_ctx = static_cast( phi::DeviceContextPool::Instance().Get(CPUPlace())); out->Resize(in.dims()); out->set_type(in.dtype()); out->set_layout(in.layout()); paddle::experimental::CheckAndTrans2Contiguous(const_cast(&in)); phi::AssignKernel(*cpu_ctx, in, out); } void Transpose2D(DenseTensor* in, DenseTensor* out) { paddle::experimental::CheckAndTrans2Contiguous(in); auto in_dims = in->dims(); PADDLE_ENFORCE_EQ( in_dims.size(), 2, common::errors::InvalidArgument( "In dims rank should be 2, but received in dims size is [%d].", in_dims.size())); DenseTensor trans_tensor; DenseTensor* out_ptr = out == nullptr ? &trans_tensor : out; out_ptr->Resize({in_dims[1], in_dims[0]}); out_ptr->set_type(in->type()); out_ptr->set_layout(in->layout()); auto* cpu_ctx = static_cast( phi::DeviceContextPool::Instance().Get(CPUPlace())); std::vector axis{1, 0}; switch (in->dtype()) { case DataType::FLOAT16: phi::TransposeKernel(*cpu_ctx, *in, axis, out_ptr); break; case DataType::FLOAT32: phi::TransposeKernel(*cpu_ctx, *in, axis, out_ptr); break; case DataType::INT16: phi::TransposeKernel(*cpu_ctx, *in, axis, out_ptr); break; case DataType::INT8: phi::TransposeKernel(*cpu_ctx, *in, axis, out_ptr); break; default: PADDLE_THROW(common::errors::InvalidArgument( "Only support fp16/fp32/int16/int8, but received dtype is %s.", DataTypeToString(in->dtype()))); break; } if (out == nullptr) { Assign(*out_ptr, in); } } void CastToInt32(DenseTensor* in, DenseTensor* out) { auto* cpu_ctx = static_cast( phi::DeviceContextPool::Instance().Get(CPUPlace())); DenseTensor int32_tensor; DenseTensor* out_ptr = out == nullptr ? &int32_tensor : out; out_ptr->Resize(in->dims()); out_ptr->set_type(DataType::INT32); out_ptr->set_layout(in->layout()); switch (in->dtype()) { case DataType::INT64: phi::CastKernel(*cpu_ctx, *in, DataType::INT32, out_ptr); break; case DataType::INT32: if (out == nullptr) { return; } else { phi::AssignKernel(*cpu_ctx, *in, out_ptr); } break; default: PADDLE_THROW(common::errors::InvalidArgument( "Only support int64 and int32, but received dtype is %s.", DataTypeToString(in->dtype()))); break; } if (out == nullptr) { Assign(*out_ptr, in); } } void CastTo(DenseTensor* in, DenseTensor* out, DataType out_dtype) { auto* cpu_ctx = static_cast( phi::DeviceContextPool::Instance().Get(CPUPlace())); if (in->dtype() != DataType::FLOAT16 && in->dtype() != DataType::FLOAT32) { PADDLE_THROW(common::errors::InvalidArgument( "Only support fp16 and fp32, but received dtype is %s.", DataTypeToString(in->dtype()))); } paddle::experimental::CheckAndTrans2Contiguous(in); DenseTensor ori_tensor; DenseTensor* out_ptr = out == nullptr ? &ori_tensor : out; out_ptr->Resize(in->dims()); out_ptr->set_type(out_dtype); out_ptr->set_layout(in->layout()); if (in->dtype() == out_dtype) { if (out == nullptr) { return; } else { phi::AssignKernel(*cpu_ctx, *in, out_ptr); } } else { if (in->dtype() == DataType::FLOAT16) { phi::CastKernel(*cpu_ctx, *in, out_dtype, out_ptr); } else { phi::CastKernel(*cpu_ctx, *in, out_dtype, out_ptr); } if (out == nullptr) { Assign(*out_ptr, in); } } } void CastToFp32(DenseTensor* in, DenseTensor* out) { CastTo(in, out, DataType::FLOAT32); } void CastToFp16(DenseTensor* in, DenseTensor* out) { CastTo(in, out, DataType::FLOAT16); } static float FindMaxAbs(const float* data, int64_t len) { float max_f = 0.0f; for (int64_t i = 0; i < len; ++i) { float max = std::abs(data[i]); if (max > max_f) { max_f = max; } } return max_f; } static float IEEECompliance0(float f) { uint32_t* ptr = reinterpret_cast(&f); uint32_t sign = (*ptr) & 0x80000000; uint32_t uf = 0; // nan -> inf if (std::isnan(f)) { uf = (sign | 0x7F800000); float* ptr = reinterpret_cast(&uf); return *ptr; } else if (std::isnormal(f) || (std::isinf(f)) || (f == 0)) { return f; } else { // denormal -> +-0 uf = 0x0; float* ptr = reinterpret_cast(&uf); return *ptr; } } static inline long RoundHalfToEven(const float src) { // NOLINT long ret = llround(src); // NOLINT if (fabs(fabs(round(src) - src) - 0.5) > 0) { return ret; } else { if (abs(ret) % 2 == 0) { return ret; } else { return ret + (ret > 0 ? -1 : 1); } } } template static T Fp32ToIntx(const float f, float max) { max = IEEECompliance0(max); float input = IEEECompliance0(f); // +0 and -0 -> +0 if (input == 0) { input = 0.0f; } float tmp = RMAX / max; if (std::isinf(tmp)) { uint32_t* ptr = reinterpret_cast(&input); if ((*ptr) >> 31 & 1) { return T(-RMAX); } else { return T(RMAX); } } tmp = input * tmp; if (std::isnan(tmp)) { return T(RMAX); } tmp = IEEECompliance0(tmp); // early check to avoid INF or big value get into convertor func. if (tmp > RMAX) { return T(RMAX); } if (tmp < -RMAX) { return T(-RMAX); } T ret = (T)RoundHalfToEven(tmp); if (ret > RMAX) { ret = T(RMAX); } if (ret < -RMAX) { ret = T(-RMAX); } return ret; } template static void QuantFP32ToIntX(const float* src_ptr, T* dst_ptr, float max_val, int numel) { PADDLE_THROW(common::errors::Unimplemented("Not support.")); } template <> void QuantFP32ToIntX(const float* src_ptr, float* dst_ptr, float max_val, int numel) { for (int i = 0; i < numel; i++) { dst_ptr[i] = static_cast(src_ptr[i]); } } template <> void QuantFP32ToIntX(const float* src_ptr, int16_t* dst_ptr, float max_val, int numel) { for (int i = 0; i < numel; i++) { dst_ptr[i] = Fp32ToIntx(src_ptr[i], max_val); } } template <> void QuantFP32ToIntX(const float* src_ptr, int8_t* dst_ptr, float max_val, int numel) { for (int i = 0; i < numel; i++) { dst_ptr[i] = Fp32ToIntx(src_ptr[i], max_val); } } template < typename Tcpu, typename Txpu, typename std::enable_if::value, Tcpu>::type* ptr> void ConvertWithQuant(DenseTensor* weight, DenseTensor* weight_max, DenseTensor* scale_max, bool transpose, bool per_channel_quant) { std::stringstream ss; ss << "Not support for Tcpu is " << phi::CppTypeToDataType::Type(); PADDLE_THROW(common::errors::Fatal(ss.str())); } template < typename Tcpu, typename Txpu, typename std::enable_if::value, Tcpu>::type* ptr> void ConvertWithQuant(DenseTensor* weight, DenseTensor* weight_max, DenseTensor* scale_max, bool transpose, bool per_channel_quant) { // Convert fp16 to fp32 DenseTensor weight_fp32; CastToFp32(weight, &weight_fp32); if (transpose) { // (k, n) -> (n, k) Transpose2D(&weight_fp32); } auto* cpu_ctx = static_cast( phi::DeviceContextPool::Instance().Get(CPUPlace())); if (!per_channel_quant) { // Find max int max_ptr_size = phi::backends::xpu::get_xpu_max_ptr_size(-1); weight_max->set_type(DataType::FLOAT32); weight_max->Resize({max_ptr_size}); int64_t size = weight_fp32.numel(); auto* weight_fp32_data = weight_fp32.data(); float max_val = FindMaxAbs(weight_fp32_data, size); std::vector max_vec(max_ptr_size, max_val); memcpy(cpu_ctx->Alloc(weight_max), max_vec.data(), max_ptr_size * sizeof(float)); // Quant weight->set_type(phi::CppTypeToDataType::Type()); weight->Resize(weight_fp32.dims()); QuantFP32ToIntX( weight_fp32_data, cpu_ctx->Alloc(weight), max_val, size); } else { std::vector quant_scales{}; auto GetQuantScales = [&](const float* weight_data, int n, int data_count) -> std::vector { std::vector scales; for (int i = 0; i < n; ++i) { float max_val = FindMaxAbs(weight_data + i * data_count, data_count); scales.push_back(max_val); } return scales; }; int64_t n = weight_fp32.dims()[0]; int64_t data_count = weight_fp32.numel() / n; auto* weight_fp32_data = weight_fp32.data(); // TODO(large-tensor): GetQuantScales and QuantFP32ToIntX not support int64 PADDLE_ENFORCE_LE_INT_MAX(n, "n"); PADDLE_ENFORCE_LE_INT_MAX(data_count, "data_count"); quant_scales = GetQuantScales( weight_fp32_data, static_cast(n), static_cast(data_count)); weight->set_type(phi::CppTypeToDataType::Type()); weight->Resize(weight_fp32.dims()); auto* weight_data = cpu_ctx->Alloc(weight); for (int64_t i = 0; i < n; ++i) { QuantFP32ToIntX(weight_fp32_data + i * data_count, weight_data + i * data_count, quant_scales[i], static_cast(data_count)); } int max_ptr_size = phi::backends::xpu::get_xpu_max_ptr_size(-1); // 1. Create weight_max tensor(all data is 1.0f) weight_max->set_type(DataType::FLOAT32); weight_max->Resize({max_ptr_size}); std::vector ones_vec(max_ptr_size, 1.0); memcpy(cpu_ctx->Alloc(weight_max), ones_vec.data(), max_ptr_size * sizeof(float)); // 2. Create scale_max tensor scale_max->set_type(DataType::FLOAT32); scale_max->Resize({static_cast(quant_scales.size())}); memcpy(cpu_ctx->Alloc(scale_max), quant_scales.data(), quant_scales.size() * sizeof(float)); } } template void ConvertWithoutQuant(DenseTensor* weight, DenseTensor* weight_max, DenseTensor* scale_max, bool transpose, const std::vector& weight_scales) { if (transpose) { Transpose2D(weight); } bool per_tensor_quant = weight_scales.size() == 1; if (std::is_same::value || std::is_same::value) { PADDLE_ENFORCE_EQ( weight_scales.empty(), false, common::errors::InvalidArgument( "ConvertWithoutQuant is not allowed weight scales is empty!")); auto* cpu_ctx = static_cast( phi::DeviceContextPool::Instance().Get(CPUPlace())); if (per_tensor_quant) { int max_ptr_size = phi::backends::xpu::get_xpu_max_ptr_size(-1); std::vector max_vec(max_ptr_size, weight_scales[0]); weight_max->set_type(DataType::FLOAT32); weight_max->Resize({max_ptr_size}); memcpy(cpu_ctx->Alloc(weight_max), max_vec.data(), max_ptr_size * sizeof(float)); } else { int max_ptr_size = phi::backends::xpu::get_xpu_max_ptr_size(-1); // 1. Create weight_max tensor(all data is 1.0f) weight_max->set_type(DataType::FLOAT32); weight_max->Resize({max_ptr_size}); std::vector ones_vec(max_ptr_size, 1.0); memcpy(cpu_ctx->Alloc(weight_max), ones_vec.data(), max_ptr_size * sizeof(float)); // 2. Create scale_max tensor scale_max->set_type(DataType::FLOAT32); scale_max->Resize({static_cast(weight_scales.size())}); memcpy(cpu_ctx->Alloc(scale_max), weight_scales.data(), weight_scales.size() * sizeof(float)); } } else if (std::is_same::value) { // Convert fp16 to fp32 DenseTensor weight_fp32; CastToFp32(weight, &weight_fp32); // Find max int max_ptr_size = phi::backends::xpu::get_xpu_max_ptr_size(-1); int64_t size = weight_fp32.numel(); auto* weight_data = weight_fp32.data(); float max_val = FindMaxAbs(weight_data, size); std::vector max_vec(max_ptr_size, max_val); weight_max->set_type(DataType::FLOAT32); weight_max->Resize({max_ptr_size}); auto* cpu_ctx = static_cast( phi::DeviceContextPool::Instance().Get(CPUPlace())); memcpy(cpu_ctx->Alloc(weight_max), max_vec.data(), max_ptr_size * sizeof(float)); // Quant weight->set_type(DataType::FLOAT32); weight->Resize(weight_fp32.dims()); QuantFP32ToIntX( weight_data, cpu_ctx->Alloc(weight), max_val, size); } else { PADDLE_THROW(common::errors::InvalidArgument( "Only support float<->int31, int8<->int8 and int16<->int16 convert.")); } } template void ConvertWithQuant(DenseTensor* weight, DenseTensor* weight_max, DenseTensor* scale_max, bool transpose, bool per_channel_quant); template void ConvertWithQuant(DenseTensor* weight, DenseTensor* weight_max, DenseTensor* scale_max, bool transpose, bool per_channel_quant); template void ConvertWithoutQuant( DenseTensor* weight, DenseTensor* weight_max, DenseTensor* scale_max, bool transpose, const std::vector& weight_scales); template void ConvertWithoutQuant( DenseTensor* weight, DenseTensor* weight_max, DenseTensor* scale_max, bool transpose, const std::vector& weight_scales); bool IsPerTensorQuant(const std::vector& weight_max) { bool per_tensor = true; PADDLE_ENFORCE_GT( weight_max.size(), 0, common::errors::InvalidArgument( "Op's channel size: [%d] should great than zero", weight_max.size())); auto first = weight_max[0]; for (size_t i = 1; i < weight_max.size(); ++i) { if (std::abs(first - weight_max[i]) > 1e-6) { per_tensor = false; break; } } return per_tensor; } } // namespace ir } // namespace framework } // namespace paddle