已关闭
MicroAPI替换为Reg(mamba/causal_conv1d)及迁移指导文档同步 #11280
景明创建于 9月4日关闭于 9月8日
MicroAPI替换为Reg(mamba/causal_conv1d)及迁移指导文档同步 #11280
已关闭
共 2 个文件变更+34-29
| @@ -209,18 +209,20 @@ Ascend 950系列引入了Regbase编程范式,相比传统的Membase(Vector A | |||
| 209 | 209 | ||
| 210 | **特点** | 210 | **特点** |
| 211 | 211 | ||
| 212 | -- 使用`AscendC::MicroAPI`命名空间下的底层API | 212 | +- 使用`AscendC::Reg`命名空间下的底层API |
| 213 | - 直接操作寄存器`RegTensor<T>`而非显式管理UB缓冲队列 | 213 | - 直接操作寄存器`RegTensor<T>`而非显式管理UB缓冲队列 |
| 214 | - 通过`MaskReg`实现灵活的元素级掩码控制 | 214 | - 通过`MaskReg`实现灵活的元素级掩码控制 |
| 215 | 215 | ||
| 216 | +> 注:`MicroAPI` 为 `Reg` 的历史别名(CANN 头文件中定义 `namespace MicroAPI = Reg;`,仅在部分新架构生效)。新代码统一使用 `AscendC::Reg` 命名空间,本仓已全量完成 MicroAPI→Reg 整改。 | ||
| 217 | + | ||
| 216 | **与Membase编程模型对比** | 218 | **与Membase编程模型对比** |
| 217 | 219 | ||
| 218 | -| 特性 | Membase(传统Vector API) | Regbase(MicroAPI) | | 220 | +| 特性 | Membase(传统Vector API) | Regbase(Reg) | |
| 219 | |------|---------------------------|---------------------| | 221 | |------|---------------------------|---------------------| |
| 220 | | 数据载体 | `LocalTensor<T>` + Queue机制 | `RegTensor<T>`寄存器 | | 222 | | 数据载体 | `LocalTensor<T>` + Queue机制 | `RegTensor<T>`寄存器 | |
| 221 | | 内存管理 | 显式Alloc/EnQue/DeQue/Free | 寄存器自动分配 | | 223 | | 内存管理 | 显式Alloc/EnQue/DeQue/Free | 寄存器自动分配 | |
| 222 | | 掩码控制 | 函数参数控制 | `MaskReg`寄存器控制 | | 224 | | 掩码控制 | 函数参数控制 | `MaskReg`寄存器控制 | |
| 223 | -| 数据搬运 | `DataCopy`/`DataCopyPad` | `MicroAPI::DataCopy` + 分发模式 | | 225 | +| 数据搬运 | `DataCopy`/`DataCopyPad` | `Reg::DataCopy` + 分发模式 | |
| 224 | 226 | ||
| 225 | **代码示例** | 227 | **代码示例** |
| 226 | 228 | ||
| @@ -228,26 +230,26 @@ Ascend 950系列引入了Regbase编程范式,相比传统的Membase(Vector A | |||
| 228 | __simd_vf__ __aicore__ void GenIndexBuf(ubuf int32_t* helpAddr, int32_t colFactor) | 230 | __simd_vf__ __aicore__ void GenIndexBuf(ubuf int32_t* helpAddr, int32_t colFactor) |
| 229 | { | 231 | { |
| 230 | // 声明寄存器张量 | 232 | // 声明寄存器张量 |
| 231 | - AscendC::MicroAPI::RegTensor<int32_t> v0; | 233 | + AscendC::Reg::RegTensor<int32_t> v0; |
| 232 | - AscendC::MicroAPI::RegTensor<int32_t> v1; | 234 | + AscendC::Reg::RegTensor<int32_t> v1; |
| 233 | - AscendC::MicroAPI::RegTensor<int32_t> vd1; | 235 | + AscendC::Reg::RegTensor<int32_t> vd1; |
| 234 | - AscendC::MicroAPI::RegTensor<int32_t> vd2; | 236 | + AscendC::Reg::RegTensor<int32_t> vd2; |
| 235 | - AscendC::MicroAPI::RegTensor<int32_t> vd3; | 237 | + AscendC::Reg::RegTensor<int32_t> vd3; |
| 236 | 238 | ||
| 237 | // 创建全量掩码 | 239 | // 创建全量掩码 |
| 238 | - AscendC::MicroAPI::MaskReg preg = | 240 | + AscendC::Reg::MaskReg preg = |
| 239 | - AscendC::MicroAPI::CreateMask<int32_t, AscendC::MicroAPI::MaskPattern::ALL>(); | 241 | + AscendC::Reg::CreateMask<int32_t, AscendC::Reg::MaskPattern::ALL>(); |
| 240 | 242 | ||
| 241 | // 标量复制到寄存器 | 243 | // 标量复制到寄存器 |
| 242 | - AscendC::MicroAPI::Duplicate(v1, colFactor, preg); | 244 | + AscendC::Reg::Duplicate(v1, colFactor, preg); |
| 243 | // 生成序列 [0, 1, 2, ...] | 245 | // 生成序列 [0, 1, 2, ...] |
| 244 | - AscendC::MicroAPI::Arange(v0, 0); | 246 | + AscendC::Reg::Arange(v0, 0); |
| 245 | // 向量运算 | 247 | // 向量运算 |
| 246 | - AscendC::MicroAPI::Div(vd1, v0, v1, preg); | 248 | + AscendC::Reg::Div(vd1, v0, v1, preg); |
| 247 | - AscendC::MicroAPI::Mul(vd2, vd1, v1, preg); | 249 | + AscendC::Reg::Mul(vd2, vd1, v1, preg); |
| 248 | - AscendC::MicroAPI::Sub(vd3, v0, vd2, preg); | 250 | + AscendC::Reg::Sub(vd3, v0, vd2, preg); |
| 249 | // 寄存器数据写回UB | 251 | // 寄存器数据写回UB |
| 250 | - AscendC::MicroAPI::DataCopy(helpAddr, vd3, preg); | 252 | + AscendC::Reg::DataCopy(helpAddr, vd3, preg); |
| 251 | } | 253 | } |
| 252 | ``` | 254 | ``` |
| 253 | 255 | ||
| @@ -255,19 +257,19 @@ __simd_vf__ __aicore__ void GenIndexBuf(ubuf int32_t* helpAddr, int32_t colFacto | |||
| 255 | // 动态掩码:处理尾部不完整数据 | 257 | // 动态掩码:处理尾部不完整数据 |
| 256 | __simd_vf__ __aicore__ void GatherProcess(ubuf int8_t* curXAddr, ubuf int8_t* curYAddr, uint16_t repeatTimes, uint16_t computeSize) | 258 | __simd_vf__ __aicore__ void GatherProcess(ubuf int8_t* curXAddr, ubuf int8_t* curYAddr, uint16_t repeatTimes, uint16_t computeSize) |
| 257 | { | 259 | { |
| 258 | - MicroAPI::RegTensor<int8_t> vregTemp; | 260 | + Reg::RegTensor<int8_t> vregTemp; |
| 259 | - MicroAPI::MaskReg preg; | 261 | + Reg::MaskReg preg; |
| 260 | // sreg为剩余待处理元素(普通unit32_t标量计数) | 262 | // sreg为剩余待处理元素(普通unit32_t标量计数) |
| 261 | unit32_t sreg = static_cast<unit32_t>(repeatimes)*computeSize; | 263 | unit32_t sreg = static_cast<unit32_t>(repeatimes)*computeSize; |
| 262 | 264 | ||
| 263 | for (uint16_t r = 0; r < repeatTimes; r++) { | 265 | for (uint16_t r = 0; r < repeatTimes; r++) { |
| 264 | // 根据剩余元素数更新掩码 | 266 | // 根据剩余元素数更新掩码 |
| 265 | - preg = MicroAPI::UpdateMask<int8_t>(sreg); | 267 | + preg = Reg::UpdateMask<int8_t>(sreg); |
| 266 | // 创建地址偏移寄存器 | 268 | // 创建地址偏移寄存器 |
| 267 | - MicroAPI::AddrReg offset = MicroAPI::CreateAddrReg<int8_t>(r, computeSize); | 269 | + Reg::AddrReg offset = Reg::CreateAddrReg<int8_t>(r, computeSize); |
| 268 | - MicroAPI::DataCopy(vregTemp, curXAddr, offset); | 270 | + Reg::DataCopy(vregTemp, curXAddr, offset); |
| 269 | // 带掩码的数据存储 | 271 | // 带掩码的数据存储 |
| 270 | - MicroAPI::DataCopy(curYAddr, vregTemp, offset, preg); | 272 | + Reg::DataCopy(curYAddr, vregTemp, offset, preg); |
| 271 | } | 273 | } |
| 272 | } | 274 | } |
| 273 | ``` | 275 | ``` |
| @@ -276,16 +278,16 @@ __simd_vf__ __aicore__ void GatherProcess(ubuf int8_t* curXAddr, ubuf int8_t* cu | |||
| 276 | // 数据聚合 | 278 | // 数据聚合 |
| 277 | __VEC_SCOPE__ | 279 | __VEC_SCOPE__ |
| 278 | { | 280 | { |
| 279 | - MicroAPI::RegTensor<uint32_t> indicesReg; | 281 | + Reg::RegTensor<uint32_t> indicesReg; |
| 280 | - MicroAPI::RegTensor<int32_t> vd0; | 282 | + Reg::RegTensor<int32_t> vd0; |
| 281 | 283 | ||
| 282 | for (uint16_t indices = 0; indices < indicesLoopNum; indices++) { | 284 | for (uint16_t indices = 0; indices < indicesLoopNum; indices++) { |
| 283 | // 加载索引(E2B分发模式:将标量广播到向量) | 285 | // 加载索引(E2B分发模式:将标量广播到向量) |
| 284 | - MicroAPI::DataCopy<uint32_t, MicroAPI::LoadDist::DIST_E2B_B32>(indicesReg, indicesAddr); | 286 | + Reg::DataCopy<uint32_t, Reg::LoadDist::DIST_E2B_B32>(indicesReg, indicesAddr); |
| 285 | // 根据索引进行Gather数据聚合 | 287 | // 根据索引进行Gather数据聚合 |
| 286 | - MicroAPI::DataCopyGather(vd0, curXAddr, indicesReg, preg); | 288 | + Reg::DataCopyGather(vd0, curXAddr, indicesReg, preg); |
| 287 | // 数据块拷贝输出 | 289 | // 数据块拷贝输出 |
| 288 | - MicroAPI::DataCopy<int32_t, MicroAPI::DataCopyMode::DATA_BLOCK_COPY>( | 290 | + Reg::DataCopy<int32_t, Reg::DataCopyMode::DATA_BLOCK_COPY>( |
| 289 | curYAddr, vd0, blockStride, preg); | 291 | curYAddr, vd0, blockStride, preg); |
| 290 | } | 292 | } |
| 291 | } | 293 | } |