// Copyright (c) 2023 PaddlePaddle Authors. All Rights Reserved. // // Licensed under the Apache License, Version 2.0 (the "License"); // you may not use this file except in compliance with the License. // You may obtain a copy of the License at // // http://www.apache.org/licenses/LICENSE-2.0 // // Unless required by applicable law or agreed to in writing, software // distributed under the License is distributed on an "AS IS" BASIS, // WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. // See the License for the specific language governing permissions and // limitations under the License. #include "paddle/phi/kernels/gather_nd_grad_kernel.h" #include "paddle/phi/backends/xpu/enforce_xpu.h" #include "paddle/phi/backends/xpu/xpu_context.h" #include "paddle/phi/core/kernel_registry.h" namespace phi { template void GatherNdGradKernel(const Context &dev_ctx, const DenseTensor &x, const DenseTensor &index, const DenseTensor &out_grad, DenseTensor *x_grad) { using XPUType = typename XPUTypeTrait::Type; dev_ctx.template Alloc(x_grad); if (x_grad->numel() == 0) { return; } int r = 0; XPUType *dx_data = reinterpret_cast(x_grad->data()); r = xpu::constant( dev_ctx.x_context(), dx_data, x_grad->numel(), static_cast(0)); PADDLE_ENFORCE_XDNN_SUCCESS(r, "constant"); if (out_grad.numel() == 0) { return; } if (index.numel() == 0) { auto index_dims = index.dims(); auto index_dims_size = index_dims.size(); // final dim int64_t end_size = index_dims[index_dims_size - 1]; PADDLE_ENFORCE_EQ( end_size, 0, common::errors::InvalidArgument("end_size[%d] should be 0", end_size)); // remain dim auto remain_ddim = slice_ddim(index_dims, 0, index_dims_size - 1); int64_t remain_numel = common::product(remain_ddim); int64_t x_numel = x.numel(); int64_t out_grad_numel = out_grad.numel(); PADDLE_ENFORCE_EQ( x_numel * remain_numel, out_grad_numel, common::errors::InvalidArgument( "x_numel[%d] * remain_numel[%d] should match out_grad_numel[%d]", x_numel, remain_numel, out_grad_numel)); // int reduce_sum(Context* xpu_ctx, const T* x, T* y, const // std::vector& xshape, const std::vector& rdims) int r = xpu::reduce_sum(dev_ctx.x_context(), reinterpret_cast(out_grad.data()), reinterpret_cast(x_grad->data()), {(int64_t)remain_numel, (int64_t)x_numel}, {0LL}); PADDLE_ENFORCE_XDNN_SUCCESS(r, "reduce_sum"); return; } auto index_type = index.dtype(); bool index_type_match = index_type == DataType::INT32 || index_type == DataType::INT64; PADDLE_ENFORCE_EQ(index_type_match, true, common::errors::InvalidArgument( "Index holds the wrong type, it holds [%s]," "but desires to be [%s] or [%s]", index_type, DataType::INT32, DataType::INT64)); auto x_shape = vectorize(x_grad->dims()); auto index_shape = vectorize(index.dims()); if (index_shape.size() == 1) { index_shape.insert(index_shape.begin(), 1); } xpu::VectorParam x_vec = { x_shape.data(), static_cast(x_shape.size()), nullptr}; int64_t index_size = index.numel(); if (index_type == DataType::INT32) { auto index_data = const_cast(index.data()); xpu::VectorParam index_vec{nullptr, index_size, index_data}; r = xpu::scatter_nd( dev_ctx.x_context(), nullptr, reinterpret_cast(out_grad.data()), dx_data, index_vec, x_vec, index_shape, false); } else { auto index_data = const_cast(index.data()); xpu::VectorParam index_vec{nullptr, index_size, index_data}; r = xpu::scatter_nd( dev_ctx.x_context(), nullptr, reinterpret_cast(out_grad.data()), dx_data, index_vec, x_vec, index_shape, false); } PADDLE_ENFORCE_XDNN_SUCCESS(r, "scatter_nd"); } } // namespace phi PD_REGISTER_KERNEL(gather_nd_grad, XPU, ALL_LAYOUT, phi::GatherNdGradKernel, float, int, phi::float16, phi::bfloat16, int64_t) {}