将输入按照输出shape进行广播。
比如A的shape为(2,1),广播的目标shape为(2,16),则会将原来的一列扩展为相同的16列。
输入数据: [[ 1] [ 2]] 输出数据: [[ 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1] [ 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2]]
以float类型,ND格式,[m, 1]广播到[m, k]为例,描述BroadCast高阶API内部算法框图,如下图所示。
计算过程分为如下几步,均在Vector上进行:
template <typename T, int32_t dim, int32_t axis, bool isReuseSource = false> __aicore__ inline void BroadCast(LocalTensor<T> &dstLocal, const LocalTensor<T> &srcLocal, const uint32_t dstShape[dim], const uint32_t srcShape[dim], LocalTensor<uint8_t> &sharedTmpBuffer)
template <typename T, int32_t dim, int32_t axis, bool isReuseSource = false> __aicore__ inline void BroadCast(LocalTensor<T> &dstLocal, const LocalTensor<T> &srcLocal, const uint32_t dstShape[dim], const uint32_t srcShape[dim])
该接口需要额外的临时空间来存储计算过程中的中间变量。临时空间支持开发者通过sharedTmpBuffer入参传入和接口框架申请两种方式。
通过sharedTmpBuffer传入的情况,开发者需要为tensor申请空间;接口框架申请的方式,开发者需要预留临时空间。临时空间大小BufferSize的获取方式如下:通过GetBroadCastMaxMinTmpSize中提供的接口获取需要预留空间范围的大小。
|
接口 |
功能 |
|---|---|
|
T |
数据类型: uint8_t/int8_t/half/float |
|
dim |
输入/输出tensor的维度,目前仅支持1维和2维 |
|
axis |
要广播的维度,目前仅支持0和1 |
|
isReuseSource |
是否允许修改源操作数。该参数预留,传入默认值false即可。 |
|
参数名 |
输入/输出 |
描述 |
|---|---|---|
|
dstTensor |
输出 |
目的操作数。 类型为LocalTensor,支持的TPosition为VECIN/VECCALC/VECOUT。 Atlas推理系列产品AI Core,支持的数据类型为:uint8_t/int8_t/half/float Atlas A2训练系列产品/Atlas 800I A2推理产品,支持的数据类型为:uint8_t/int8_t/half/float |
|
srcTensor |
输入 |
源操作数。 源操作数的数据类型需要与目的操作数保持一致。 类型为LocalTensor,支持的TPosition为VECIN/VECCALC/VECOUT。 Atlas推理系列产品AI Core,支持的数据类型为:uint8_t/int8_t/half/float Atlas A2训练系列产品/Atlas 800I A2推理产品,支持的数据类型为:uint8_t/int8_t/half/float |
|
dstShape |
输入 |
输出tensor的shape:uint32_t 类型的数组,长度为1或者2, 输入/输出的shape维度要一致。 |
|
srcShape |
输入 |
输入tensor的shape:uint32_t 类型的数组,长度为1或者2, 输入/输出的shape维度要一致。 |
|
sharedTmpBuffer |
输入 |
临时缓存。 类型为LocalTensor,支持的TPosition为VECIN/VECCALC/VECOUT。 用于Broadcast内部复杂计算时存储中间变量,由开发者提供。 临时空间大小BufferSize的获取方式请参考BroadCast Tiling。 |
无
Atlas推理系列产品AI Core
Atlas A2训练系列产品/Atlas 800I A2推理产品
#ifdef ASCENDC_CPU_DEBUG
#include "tikicpulib.h"
#include <cstdlib>
#endif
#include "kernel_operator.h"
using namespace AscendC;
template <typename T, int32_t dim, int32_t axis>
class KernelBroadCast {
public:
__aicore__ inline KernelBroadCast()
{}
__aicore__ inline void Init(GM_ADDR src_gm, GM_ADDR dst_gm, const uint32_t dstShape[dim],
const uint32_t srcShape[dim])
{
for (uint32_t i = 0; i < dim; i++) {
srcSize *= srcShape[i];
dstSize *= dstShape[i];
}
src_global.SetGlobalBuffer(reinterpret_cast<__gm__ T *>(src_gm), srcSize);
dst_global.SetGlobalBuffer(reinterpret_cast<__gm__ T *>(dst_gm), dstSize);
pipe.InitBuffer(inQueueX, 1, srcSize * sizeof(T));
pipe.InitBuffer(outQueue, 1, dstSize * sizeof(T));
dstShape_ = dstShape;
srcShape_ = srcShape;
}
__aicore__ inline void Process()
{
CopyIn();
Compute();
CopyOut();
}
private:
__aicore__ inline void CopyIn()
{
LocalTensor<T> srcLocal = inQueueX.AllocTensor<T>();
DataCopy(srcLocal, src_global, srcSize);
inQueueX.EnQue(srcLocal);
}
__aicore__ inline void Compute()
{
LocalTensor<T> dstLocal = outQueue.AllocTensor<T>();
LocalTensor<T> srcLocal = inQueueX.DeQue<T>();
BroadCast<T, dim, axis>(dstLocal, srcLocal, dstShape_, srcShape_);
outQueue.EnQue<T>(dstLocal);
inQueueX.FreeTensor(srcLocal);
}
__aicore__ inline void CopyOut()
{
LocalTensor<T> dstLocal = outQueue.DeQue<T>();
DataCopy(dst_global, dstLocal, dstSize);
outQueue.FreeTensor(dstLocal);
}
private:
GlobalTensor<T> src_global;
GlobalTensor<T> dst_global;
TPipe pipe;
TQue<QuePosition::VECIN, 1> inQueueX;
TQue<QuePosition::VECOUT, 1> outQueue;
const uint32_t *dstShape_{nullptr};
const uint32_t *srcShape_{nullptr};
int32_t srcSize{1};
int32_t dstSize{1};
};
template <typename T, int32_t dim, int32_t axis>
__aicore__ void kernel_broadcast_operator(GM_ADDR src_gm, GM_ADDR dst_gm, const uint32_t dstShape[dim], const uint32_t srcShape[dim])
{
KernelBroadCast <T,dim, axis> op;
op.Init(src_gm, dst_gm, dstShape, srcShape);
op.Process();
}
输入数据(srcTensor): [[ 1] [ 2] [ 3] [ 4] [ 5] [ 6] [ 7] [ 8] [ 9] [10] [11] [12] [13] [14] [15] [16]] 输出数据(dstLocal): [[ 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1] [ 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2] [ 3 3 3 3 3 3 3 3 3 3 3 3 3 3 3 3] [ 4 4 4 4 4 4 4 4 4 4 4 4 4 4 4 4] [ 5 5 5 5 5 5 5 5 5 5 5 5 5 5 5 5] [ 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6] [ 7 7 7 7 7 7 7 7 7 7 7 7 7 7 7 7] [ 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8] [ 9 9 9 9 9 9 9 9 9 9 9 9 9 9 9 9] [10 10 10 10 10 10 10 10 10 10 10 10 10 10 10 10] [11 11 11 11 11 11 11 11 11 11 11 11 11 11 11 11] [12 12 12 12 12 12 12 12 12 12 12 12 12 12 12 12] [13 13 13 13 13 13 13 13 13 13 13 13 13 13 13 13] [14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14] [15 15 15 15 15 15 15 15 15 15 15 15 15 15 15 15] [16 16 16 16 16 16 16 16 16 16 16 16 16 16 16 16]]