aclnnAminmax
Supported Products
| Product | Supported |
|---|---|
| √ | |
| √ | |
| × | |
| √ | |
| √ |
Function Description
Returns the minimum and maximum values of each row of the input tensor along the specified dimension.
Prototype
Each operator has two-phase API calls. First, aclnnAminmaxGetWorkspaceSize is called to obtain the input parameters and compute the required workspace size based on the process. Then, aclnnAminmax is called to perform computation.
aclnnStatus aclnnAminmaxGetWorkspaceSize(
const aclTensor *self,
const aclIntArray *dim,
bool keepDim,
aclTensor *minOut,
aclTensor *maxOut,
uint64_t *workspaceSize,
aclOpExecutor **executor)aclnnStatus aclnnAminmax(
void *workspace,
uint64_t workspaceSize,
aclOpExecutor *executor,
aclrtStream stream)aclnnAminmaxGetWorkspaceSize
Parameters:
Name Input/Output Description Usage Notes Data Type Data Format Dimension (Shape) Non-contiguous Tensor self Input Input tensor. - FLOAT, BFLOAT16, FLOAT16, DOUBLE, INT8, INT16, INT32, INT64, UINT8, or BOOL ND - √ dim Input Specifies the dimension to be reduced. The value range is [–self.dim(), self.dim() – 1] INT64 - - - keepdim Input Indicates whether to retain the reduction axes. - BOOL - - - minOut Output An output tensor storing the minimum values. Its data type must be the same as that of self.FLOAT, BFLOAT16, FLOAT16, DOUBLE, INT8, INT16, INT32, INT64, UINT8, or BOOL ND - √ maxOut Output An output tensor storing the maximum values. Its data type must be the same as that of self.FLOAT, BFLOAT16, FLOAT16, DOUBLE, INT8, INT16, INT32, INT64, UINT8, or BOOL ND - √ workspaceSize Output Size of the workspace required to be allocated on the device. - - - - - executor Output Operator executor, covering the operator computation process. - - - - - Atlas inference products andAtlas training products : The data type cannot be BFLOAT16.
Returns:
aclnnStatus: status code. For details, see aclnn Return Codes.The first-phase API implements input parameter verification. The following errors may be thrown.
Return Error Code Description ACLNN_ERR_PARAM_NULLPTR 161001 The input self,dim,minOut, ormaxOutis a null pointer.ACLNN_ERR_PARAM_INVALID 161002 The data type of selfis not supported.The data type of minOutormaxOutis different from that ofself.self,minOut, ormaxOuthas more than 8 dimensions.The value of dimis out of the range.The number of elements in dimis 1 and the shape of this specified dimension inselfis 0.The number of elements in dimis not 1 andselfis an empty tensor.
aclnnAminmax
Parameters:
Name Input/Output Description workspace Input Address of the workspace to be allocated on the device. workspaceSize Input Size of the workspace to be allocated on the device, obtained by calling the first-phase API aclnnAminmaxGetWorkspaceSize.executor Input Operator executor, covering the operator computation process. stream Input Stream for executing the task. Returns:
aclnnStatus: status code. For details, see aclnn Return Codes.
Constraints
- Deterministic computation:
aclnnAminmaxdefaults to a deterministic implementation.
Example
The following example is for reference only. For details, see Compilation and Running Sample.
#include <iostream>
#include <vector>
#include "acl/acl.h"
#include "aclnnop/aclnn_aminmax.h"
#define CHECK_RET(cond, return_expr) \
do { \
if (!(cond)) { \
return_expr; \
} \
} while (0)
#define LOG_PRINT(message, ...) \
do { \
printf(message, ##__VA_ARGS__); \
} while (0)
int64_t GetShapeSize(const std::vector<int64_t>& shape) {
int64_t shapeSize = 1;
for (auto i : shape) {
shapeSize *= i;
}
return shapeSize;
}
int Init(int32_t deviceId, aclrtStream* stream) {
// Boilerplate code for resource initialization.
auto ret = aclInit(nullptr);
CHECK_RET(ret == ACL_SUCCESS, LOG_PRINT("aclInit failed. ERROR: %d\n", ret); return ret);
ret = aclrtSetDevice(deviceId);
CHECK_RET(ret == ACL_SUCCESS, LOG_PRINT("aclrtSetDevice failed. ERROR: %d\n", ret); return ret);
ret = aclrtCreateStream(stream);
CHECK_RET(ret == ACL_SUCCESS, LOG_PRINT("aclrtCreateStream failed. ERROR: %d\n", ret); return ret);
return 0;
}
template <typename T>
int CreateAclTensor(const std::vector<T>& hostData, const std::vector<int64_t>& shape, void** deviceAddr,
aclDataType dataType, aclTensor** tensor) {
auto size = GetShapeSize(shape) * sizeof(T);
// Call aclrtMalloc to allocate memory on the device.
auto ret = aclrtMalloc(deviceAddr, size, ACL_MEM_MALLOC_HUGE_FIRST);
CHECK_RET(ret == ACL_SUCCESS, LOG_PRINT("aclrtMalloc failed. ERROR: %d\n", ret); return ret);
// Call aclrtMemcpy to copy the data from the host to the device.
ret = aclrtMemcpy(*deviceAddr, size, hostData.data(), size, ACL_MEMCPY_HOST_TO_DEVICE);
CHECK_RET(ret == ACL_SUCCESS, LOG_PRINT("aclrtMemcpy failed. ERROR: %d\n", ret); return ret);
// Compute the strides of the contiguous tensor.
std::vector<int64_t> strides(shape.size(), 1);
for (int64_t i = shape.size() - 2; i >= 0; i--) {
strides[i] = shape[i + 1] * strides[i + 1];
}
// Call aclCreateTensor to create an aclTensor.
*tensor = aclCreateTensor(shape.data(), shape.size(), dataType, strides.data(), 0, aclFormat::ACL_FORMAT_ND,
shape.data(), shape.size(), *deviceAddr);
return 0;
}
int main() {
// 1. Boilerplate code for device/stream initialization. For details, see the ACL API manual.
// Set the device ID in use.
int32_t deviceId = 0;
aclrtStream stream;
auto ret = Init(deviceId, &stream);
// Customize the error handling as needed.
CHECK_RET(ret == ACL_SUCCESS, LOG_PRINT("Init acl failed. ERROR: %d\n", ret); return ret);
// 2. Construct the inputs and outputs based on the API definition.
std::vector<int64_t> selfShape = {2, 3, 2};
std::vector<int64_t> outShape = {1, 3, 2};
void* selfDeviceAddr = nullptr;
void* minOutDeviceAddr = nullptr;
void* maxOutDeviceAddr = nullptr;
aclTensor* self = nullptr;
aclTensor* minOut = nullptr;
aclTensor* maxOut = nullptr;
std::vector<float> selfHostData = {0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11};
std::vector<float> minOutHostData = {0, 0, 0, 0, 0, 0};
std::vector<float> maxOutHostData = {0, 0, 0, 0, 0, 0};
std::vector<int64_t> dimData = {0};
bool keepDim = true;
// Create a self aclTensor.
ret = CreateAclTensor(selfHostData, selfShape, &selfDeviceAddr, aclDataType::ACL_FLOAT, &self);
CHECK_RET(ret == ACL_SUCCESS, return ret);
// Create a dim aclIntArray.
aclIntArray *dim = aclCreateIntArray(dimData.data(), dimData.size());
// Create an out aclTensor.
ret = CreateAclTensor(minOutHostData, outShape, &minOutDeviceAddr, aclDataType::ACL_FLOAT, &minOut);
CHECK_RET(ret == ACL_SUCCESS, return ret);
ret = CreateAclTensor(maxOutHostData, outShape, &maxOutDeviceAddr, aclDataType::ACL_FLOAT, &maxOut);
CHECK_RET(ret == ACL_SUCCESS, return ret);
// 3. Call the CANN operator library API, which needs to be replaced with the actual one.
uint64_t workspaceSize = 0;
aclOpExecutor* executor;
// Call the first-phase API of aclnnAminmax.
ret = aclnnAminmaxGetWorkspaceSize(self, dim, keepDim, minOut, maxOut, &workspaceSize, &executor);
CHECK_RET(ret == ACL_SUCCESS, LOG_PRINT("aclnnAminmaxGetWorkspaceSize failed. ERROR: %d\n", ret); return ret);
// Allocate device memory based on the workspaceSize calculated by the first-phase API.
void* workspaceAddr = nullptr;
if (workspaceSize > 0) {
ret = aclrtMalloc(&workspaceAddr, workspaceSize, ACL_MEM_MALLOC_HUGE_FIRST);
CHECK_RET(ret == ACL_SUCCESS, LOG_PRINT("allocate workspace failed. ERROR: %d\n", ret); return ret);
}
// Call the second-phase API of aclnnAminmax.
ret = aclnnAminmax(workspaceAddr, workspaceSize, executor, stream);
CHECK_RET(ret == ACL_SUCCESS, LOG_PRINT("aclnnAminmax failed. ERROR: %d\n", ret); return ret);
// 4. (Boilerplate code) Wait until the task execution is complete.
ret = aclrtSynchronizeStream(stream);
CHECK_RET(ret == ACL_SUCCESS, LOG_PRINT("aclrtSynchronizeStream failed. ERROR: %d\n", ret); return ret);
// 5. Obtain the output value and copy the result from the device to the host. Modify the code based on the API definition.
auto size = GetShapeSize(outShape);
std::vector<float> minResultData(size, 0);
ret = aclrtMemcpy(minResultData.data(), minResultData.size() * sizeof(minResultData[0]), minOutDeviceAddr,
size * sizeof(minResultData[0]), ACL_MEMCPY_DEVICE_TO_HOST);
CHECK_RET(ret == ACL_SUCCESS, LOG_PRINT("copy result from device to host failed. ERROR: %d\n", ret); return ret);
for (int64_t i = 0; i < size; i++) {
LOG_PRINT("result[%ld] is: %f\n", i, minResultData[i]);
}
std::vector<float> maxResultData(size, 0);
ret = aclrtMemcpy(maxResultData.data(), maxResultData.size() * sizeof(maxResultData[0]), maxOutDeviceAddr,
size * sizeof(maxResultData[0]), ACL_MEMCPY_DEVICE_TO_HOST);
CHECK_RET(ret == ACL_SUCCESS, LOG_PRINT("copy result from device to host failed. ERROR: %d\n", ret); return ret);
for (int64_t i = 0; i < size; i++) {
LOG_PRINT("result[%ld] is: %f\n", i, maxResultData[i]);
}
// 6. Destroy aclTensor. Modify the code based on the API definition.
aclDestroyTensor(self);
aclDestroyIntArray(dim);
aclDestroyTensor(minOut);
aclDestroyTensor(maxOut);
// 7. Free device resources. Modify the code based on the API definition.
aclrtFree(selfDeviceAddr);
aclrtFree(minOutDeviceAddr);
aclrtFree(maxOutDeviceAddr);
if (workspaceSize > 0) {
aclrtFree(workspaceAddr);
}
aclrtDestroyStream(stream);
aclrtResetDevice(deviceId);
aclFinalize();
return 0;
}