已关闭
[Bug-Report|缺陷反馈]: aclnnConfusionTranspose接口在切片场景报错 #1291
AlfengYuan创建于  4月17日关闭于  4月17日
AlfengYuan
AlfengYuan成员
4月17日 创建

Thanks for sending an issue! Please fill in the following template to help quickly solve your problem.

一、问题描述 (必填)

在 切片 tensor场景下,aclnnConfusioinTranspose接口报错,tiling测校验拦截,关键日志如下:

[ERROR] OP(1506281,opapi_test):2026-04-17-14:40:30.884.448 [../../../conversion/confusion_transpose_d/op_host/arch35/confusion_transpose_d_tiling_arch35.cpp:674][OPS_MATH][ParametersVerifyingProdAndPositive][1506281] OpName:[ConfusionTransposeD] x, output and shape must have equal dimension product, but actually 8, 9, and 8.
[INFO] GE(1506281,opapi_test):2026-04-17-14:40:30.884.484 [error_manager.cc:399]1506281 ReportInterErrMessage:report error_message, error_code:EZ9999, work_stream_id:150629606281, error_mode:0
[ERROR] OP(1506281,opapi_test):2026-04-17-14:40:30.884.496 [../../../conversion/confusion_transpose_d/op_host/arch35/confusion_transpose_d_tiling_arch35.cpp:809][OPS_MATH][ConfusionTransposeDTilingForAscendC][1506281] OpName:[ConfusionTransposeD] ConfusionTransposeDTiling failed to verify params!

二、环境信息 (可选)

ascend950
cann9.0.0

三、重现步骤 (可选)

/**
 * Copyright (c) 2026 Huawei Technologies Co., Ltd.
 * This program is free software, you can redistribute it and/or modify it under the terms and conditions of
 * CANN Open Software License Agreement Version 2.0 (the "License").
 * Please refer to the License for details. You may not use this file except in compliance with the License.
 * THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
 * INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
 * See LICENSE in the root of the software repository for the full text of the License.
 */

#include <iostream>
#include <vector>
#include "acl/acl.h"
#include "aclnnop/aclnn_confusion_transpose.h"

#define CHECK_RET(cond, return_expr) \
  do {                               \
    if (!(cond)) {                   \
      return_expr;                   \
    }                                \
  } while (0)

#define LOG_PRINT(message, ...)     \
  do {                              \
    printf(message, ##__VA_ARGS__); \
  } while (0)

int64_t GetShapeSize(const std::vector<int64_t>& shape) {
  int64_t shapeSize = 1;
  for (auto i : shape) {
    shapeSize *= i;
  }
  return shapeSize;
}

int Init(int32_t deviceId, aclrtStream* stream) {
  auto ret = aclInit(nullptr);
  CHECK_RET(ret == ACL_SUCCESS, LOG_PRINT("aclInit failed. ERROR: %d\n", ret); return ret);
  ret = aclrtSetDevice(deviceId);
  CHECK_RET(ret == ACL_SUCCESS, LOG_PRINT("aclrtSetDevice failed. ERROR: %d\n", ret); return ret);
  ret = aclrtCreateStream(stream);
  CHECK_RET(ret == ACL_SUCCESS, LOG_PRINT("aclrtCreateStream failed. ERROR: %d\n", ret); return ret);
  return 0;
}

template <typename T>
int CreateAclTensor(const std::vector<T>& hostData, const std::vector<int64_t>& shape, void** deviceAddr,
                    aclDataType dataType, aclTensor** tensor) {
  auto size = GetShapeSize(shape) * sizeof(T);
  auto ret = aclrtMalloc(deviceAddr, size, ACL_MEM_MALLOC_HUGE_FIRST);
  CHECK_RET(ret == ACL_SUCCESS, LOG_PRINT("aclrtMalloc failed. ERROR: %d\n", ret); return ret);
  ret = aclrtMemcpy(*deviceAddr, size, hostData.data(), size, ACL_MEMCPY_HOST_TO_DEVICE);
  CHECK_RET(ret == ACL_SUCCESS, LOG_PRINT("aclrtMemcpy failed. ERROR: %d\n", ret); return ret);

  std::vector<int64_t> strides(shape.size(), 1);
  for (int64_t i = shape.size() - 2; i >= 0; i--) {
    strides[i] = shape[i + 1] * strides[i + 1];
  }

  *tensor = aclCreateTensor(shape.data(), shape.size(), dataType, strides.data(), 0, aclFormat::ACL_FORMAT_ND,
                            shape.data(), shape.size(), *deviceAddr);
  return 0;
}

int main() {
  int32_t deviceId = 0;
  aclrtStream stream;
  auto ret = Init(deviceId, &stream);
  CHECK_RET(ret == ACL_SUCCESS, LOG_PRINT("Init acl failed. ERROR: %d\n", ret); return ret);

  // 2. 构造输入
  // x: shape=[2, 4], data = [1,2,3,4,5,6,7,8]
  aclTensor* x = nullptr;
  std::vector<int64_t> xShape = {2, 4};
  std::vector<float> xHostData = {1, 2, 3, 4, 5, 6, 7, 8};
  void* xDeviceAddr = nullptr;
  ret = CreateAclTensor(xHostData, xShape, &xDeviceAddr, aclDataType::ACL_FLOAT, &x);
  CHECK_RET(ret == ACL_SUCCESS, return ret);

  // perm = [1, 0]: transpose dims 0 and 1
  aclIntArray* perm = nullptr;
  std::vector<int64_t> permData = {1, 0};
  perm = aclCreateIntArray(permData.data(), permData.size());
  CHECK_RET(perm != nullptr, return ret);

  // shape = [4, 2]: output shape after transpose
  aclIntArray* shape = nullptr;
  std::vector<int64_t> shapeData = {4, 2};
  shape = aclCreateIntArray(shapeData.data(), shapeData.size());
  CHECK_RET(shape != nullptr, return ret);

  // transposeFirst = true
  bool transposeFirst = true;

  // output with offset: viewShape=[4, 2], storageShape=[9], offset=1
  // 8 view elements + 1 before offset = 9 total storage elements
  aclTensor* out = nullptr;
  void* outDeviceAddr = nullptr;
  {
    std::vector<int64_t> outViewShape = {4, 2};
    std::vector<int64_t> outStorageShape = {9};
    std::vector<int64_t> outViewStrides = {2, 1};
    int64_t outOffset = 1;
    auto size = GetShapeSize(outStorageShape) * sizeof(float);
    ret = aclrtMalloc(&outDeviceAddr, size, ACL_MEM_MALLOC_HUGE_FIRST);
    CHECK_RET(ret == ACL_SUCCESS, LOG_PRINT("aclrtMalloc for out failed. ERROR: %d\n", ret); return ret);
    out = aclCreateTensor(outViewShape.data(), outViewShape.size(), aclDataType::ACL_FLOAT,
                          outViewStrides.data(), outOffset, aclFormat::ACL_FORMAT_ND,
                          outStorageShape.data(), outStorageShape.size(), outDeviceAddr);
    CHECK_RET(out != nullptr, return ret);
  }

  // 3. 调用aclnnConfusionTranspose两段式接口
  uint64_t workspaceSize = 0;
  aclOpExecutor* executor;
  ret = aclnnConfusionTransposeGetWorkspaceSize(x, perm, shape, transposeFirst, out, &workspaceSize, &executor);
  CHECK_RET(ret == ACL_SUCCESS, LOG_PRINT("aclnnConfusionTransposeGetWorkspaceSize failed. ERROR: %d\n", ret); return ret);

  void* workspaceAddr = nullptr;
  if (workspaceSize > 0) {
    ret = aclrtMalloc(&workspaceAddr, workspaceSize, ACL_MEM_MALLOC_HUGE_FIRST);
    CHECK_RET(ret == ACL_SUCCESS, LOG_PRINT("allocate workspace failed. ERROR: %d\n", ret); return ret);
  }

  ret = aclnnConfusionTranspose(workspaceAddr, workspaceSize, executor, stream);
  CHECK_RET(ret == ACL_SUCCESS, LOG_PRINT("aclnnConfusionTranspose failed. ERROR: %d\n", ret); return ret);

  // 4. 同步等待
  ret = aclrtSynchronizeStream(stream);
  CHECK_RET(ret == ACL_SUCCESS, LOG_PRINT("aclrtSynchronizeStream failed. ERROR: %d\n", ret); return ret);

  // 5. 打印结果: 打印完整storage数据,然后打印view部分
  {
    std::vector<int64_t> outStorageShape = {9};
    std::vector<int64_t> outViewShape = {4, 2};
    int64_t outOffset = 1;
    int64_t storageElemCount = GetShapeSize(outStorageShape);
    int64_t viewElemCount = GetShapeSize(outViewShape);

    std::vector<float> outData(storageElemCount, 0);
    ret = aclrtMemcpy(outData.data(), outData.size() * sizeof(float), outDeviceAddr,
                      storageElemCount * sizeof(float), ACL_MEMCPY_DEVICE_TO_HOST);
    CHECK_RET(ret == ACL_SUCCESS, LOG_PRINT("copy result from device to host failed. ERROR: %d\n", ret); return ret);

    LOG_PRINT("Input x [2, 4]:\n");
    for (int64_t i = 0; i < 8; i++) {
      LOG_PRINT("  [%ld] = %f\n", i, xHostData[i]);
    }
    LOG_PRINT("\nOutput out: viewShape=[4, 2], storageShape=[9], offset=1, perm=[1,0], transposeFirst=true\n");
    LOG_PRINT("  Storage data (all %ld elements):\n", storageElemCount);
    for (int64_t i = 0; i < storageElemCount; i++) {
      LOG_PRINT("    [%ld] = %f%s\n", i, outData[i], i < outOffset ? " (before offset)" : "");
    }
    LOG_PRINT("  View data (%ld elements, from offset %ld):\n", viewElemCount, outOffset);
    for (int64_t i = 0; i < viewElemCount; i++) {
      LOG_PRINT("    [%ld] = %f\n", i, outData[outOffset + i]);
    }
  }

  // 6. 释放资源
  aclDestroyTensor(x);
  aclDestroyTensor(out);
  aclDestroyIntArray(perm);
  aclDestroyIntArray(shape);

  // 7. 释放device资源
  aclrtFree(xDeviceAddr);
  aclrtFree(outDeviceAddr);
  if (workspaceSize > 0) {
    aclrtFree(workspaceAddr);
  }
  aclrtDestroyStream(stream);
  aclrtResetDevice(deviceId);
  aclFinalize();
  return 0;
}
# Copyright (c) Huawei Technologies Co., Ltd. 2019. All rights reserved.

# CMake lowest version requirement
cmake_minimum_required(VERSION 3.14)

# 设置工程名
project(ACLNN_EXAMPLE)

# 设置默认的构建类型为 Debug
if(NOT CMAKE_BUILD_TYPE)
    set(CMAKE_BUILD_TYPE Debug CACHE STRING "Choose the type of build." FORCE)
endif()

# Compile options
add_compile_options(-std=c++17)

# 设置编译选项
set(CMAKE_CXX_COMPILER "g++")
set(CMAKE_RUNTIME_OUTPUT_DIRECTORY  "./bin")
set(CMAKE_CXX_FLAGS_DEBUG "-fPIC -O0 -g -Wall")
set(CMAKE_CXX_FLAGS_RELEASE "-fPIC -O2 -Wall")
set(CMAKE_SKIP_RPATH TRUE)

# 设置可执行文件名(如opapi_test),并指定待运行算子文件*.cpp所在目录
add_executable(opapi_test
               main.cpp)
# 设置ASCEND_PATH(CANN软件包目录,请根据实际路径修改)和INCLUDE_BASE_DIR(头文件目录)
if(NOT "$ENV{ASCEND_CUSTOM_PATH}" STREQUAL "")
    set(ASCEND_PATH $ENV{ASCEND_CUSTOM_PATH})
else()
    set(ASCEND_PATH "/home/developer/Ascend/cann")
endif()
set(INCLUDE_BASE_DIR "${ASCEND_PATH}/include")
message(STATUS "INCLUDE_BASE_DIR = ${INCLUDE_BASE_DIR}")
include_directories(
    ${INCLUDE_BASE_DIR}
    ${INCLUDE_BASE_DIR}/aclnn
)

# 设置链接的库文件路径
target_link_libraries(opapi_test PRIVATE
                      ${ASCEND_PATH}/lib64/libascendcl.so
                      ${ASCEND_PATH}/lib64/libnnopbase.so
                      ${ASCEND_PATH}/lib64/libopapi_nn.so
                      ${ASCEND_PATH}/lib64/libopapi_math.so
                      pthread)

# 可执行文件在CMakeLists文件所在目录的bin目录下
install(TARGETS opapi_test DESTINATION ${CMAKE_RUNTIME_OUTPUT_DIRECTORY})
source /home/developer/Ascend/cann/set_env.sh
rm -rf build && cmake -Bbuild && cmake --build build && ./build/bin/opapi_test

开启debug打屏日志
export ASCEND_GLOBAL_LOG_LEVEL=0
export ASCEND_SLOG_PRINT_TO_STDOUT=1

使用上述脚本,cmake文件, 在ascend950机器上, 在cann9.0.0环境下,编译运行上述c++代码。

四、预期结果 (可选)

预期成功运行,日志无报错。

💡 备注(选填)

likedislike
AlfengYuanAlfengYuan成员
4月17日 添加了label:bug-report
AlfengYuanAlfengYuan成员
4月17日 修改标题为 “[Bug-Report|缺陷反馈]: aclnnConfusionTranspose接口在切片场景报错拦截”,原标题为“[Bug-Report|缺陷反馈]: ”
AlfengYuanAlfengYuan成员
4月17日 修改标题为 “[Bug-Report|缺陷反馈]: aclnnConfusionTranspose接口在切片场景报错”,原标题为“[Bug-Report|缺陷反馈]: aclnnConfusionTranspose接口在切片场景报错拦截”
AlfengYuanAlfengYuan成员
4月17日 修改了issue 的描述
AlfengYuanAlfengYuan成员
4月17日 修改了issue 的描述
AlfengYuanAlfengYuan成员
4月17日 修改了issue 的描述
AlfengYuan
AlfengYuan成员
4月17日 评论:

/assign

likedislike
CANN-robotCANN-robot成员
4月17日 将 alfengyuan 设为负责人
CANN-robotCANN-robot成员
4月17日 关闭了 issue
AlfengYuanAlfengYuan成员
4月17日 issue状态由 进行中 改变为 已完成
CANN-robotCANN-robot成员
4月17日 添加了label:Accepted
CANN-robotCANN-robot成员
4月17日 添加了label:resolved