已合并
Modify the aclnn and readme issue #3367
zhouwenfang创建于 3月31日
Modify the aclnn and readme issue #3367
已合并
共 42 个文件变更+113-117
| @@ -1,3 +1,3 @@ | |||
| 1 | # AscendAntiQuantV2 | 1 | # AscendAntiQuantV2 |
| 2 | 2 | ||
| 3 | -本目录仅包含AscendAntiQuantV2算子对应的aclnn接口;如您想要贡献该算子的AscendC实现,请参考[贡献流程](../../CONTRIBUTING.md)。 | 3 | +本目录仅包含AscendAntiQuantV2算子对应的aclnn接口;如您想要贡献该算子的AscendC实现,请参考[贡献流程](../../CONTRIBUTING.md)。 |
| @@ -201,7 +201,8 @@ aclnnStatus aclnnAscendAntiQuant( | |||
| 201 | <td>dstType不在有效取值范围。</td> | 201 | <td>dstType不在有效取值范围。</td> |
| 202 | </tr> | 202 | </tr> |
| 203 | <tr> | 203 | <tr> |
| 204 | - <td>x的数据类型为INT4时,x的shape尾轴大小不是偶数。</tr> | 204 | + <td>x的数据类型为INT4时,x的shape尾轴大小不是偶数。</td> |
| 205 | + </tr> | ||
| 205 | <tr> | 206 | <tr> |
| 206 | <td>x的数据类型为INT32时,y的shape尾轴不是x的shape尾轴大小的8倍,或者x与y的shape的非尾轴的大小不一致。</td> | 207 | <td>x的数据类型为INT32时,y的shape尾轴不是x的shape尾轴大小的8倍,或者x与y的shape的非尾轴的大小不一致。</td> |
| 207 | </tr> | 208 | </tr> |
| @@ -16,13 +16,13 @@ | |||
| 16 | - 算子功能:对输入x进行量化操作,且scale和offset的size需要是x的最后一维或1。 | 16 | - 算子功能:对输入x进行量化操作,且scale和offset的size需要是x的最后一维或1。 |
| 17 | - 计算公式: | 17 | - 计算公式: |
| 18 | - sqrtMode为false时,计算公式为: | 18 | - sqrtMode为false时,计算公式为: |
| 19 | - | 19 | + |
| 20 | $$ | 20 | $$ |
| 21 | y = round((x * scale) + offset) | 21 | y = round((x * scale) + offset) |
| 22 | $$ | 22 | $$ |
| 23 | 23 | ||
| 24 | - sqrtMode为true时,计算公式为: | 24 | - sqrtMode为true时,计算公式为: |
| 25 | - | 25 | + |
| 26 | $$ | 26 | $$ |
| 27 | y = round((x * scale * scale) + offset) | 27 | y = round((x * scale * scale) + offset) |
| 28 | $$ | 28 | $$ |
| @@ -96,7 +96,7 @@ | |||
| 96 | </tr> | 96 | </tr> |
| 97 | </tbody></table> | 97 | </tbody></table> |
| 98 | 98 | ||
| 99 | -- Ascend 950PR/Ascend 950DT </term>:数据类型支持FLOAT32、FLOAT16。 | 99 | +- <term>Ascend 950PR/Ascend 950DT</term>:数据类型支持FLOAT32、FLOAT16。 |
| 100 | 100 | ||
| 101 | ## 约束说明 | 101 | ## 约束说明 |
| 102 | 102 | ||
| @@ -106,4 +106,4 @@ | |||
| 106 | 106 | ||
| 107 | | 调用方式 | 样例代码 | 说明 | | 107 | | 调用方式 | 样例代码 | 说明 | |
| 108 | | ---------------- | --------------------------- | --------------------------------------------------- | | 108 | | ---------------- | --------------------------- | --------------------------------------------------- | |
| 109 | -| 图模式 | - | 通过[算子IR](op_graph/ascend_quant_proto.h)构图方式调用AscendQuant算子。 | | 109 | +| 图模式 | - | 通过[算子IR](op_graph/ascend_quant_proto.h)构图方式调用AscendQuant算子。 | |
| @@ -121,6 +121,7 @@ | |||
| 121 | - axis:支持指定x的最后两个维度(假设输入x维度是xDimNum,axis取值范围是[-2,-1]或[xDimNum-2,xDimNum-1])。 | 121 | - axis:支持指定x的最后两个维度(假设输入x维度是xDimNum,axis取值范围是[-2,-1]或[xDimNum-2,xDimNum-1])。 |
| 122 | 122 | ||
| 123 | - Kirin X90/Kirin 9030 处理器系列产品: `x`、`scale`、`offset`不支持BFLOAT16;`y` 数据类型不支持INT4、HIFLOAT8、FLOAT8_E5M2、FLOAT8_E4M3FN。 | 123 | - Kirin X90/Kirin 9030 处理器系列产品: `x`、`scale`、`offset`不支持BFLOAT16;`y` 数据类型不支持INT4、HIFLOAT8、FLOAT8_E5M2、FLOAT8_E4M3FN。 |
| 124 | + | ||
| 124 | ## 约束说明 | 125 | ## 约束说明 |
| 125 | 126 | ||
| 126 | 无 | 127 | 无 |
| @@ -131,4 +132,4 @@ | |||
| 131 | | ---------------- | --------------------------- | --------------------------------------------------- | | 132 | | ---------------- | --------------------------- | --------------------------------------------------- | |
| 132 | | aclnn接口 | [test_aclnn_ascend_quant](examples/test_aclnn_ascend_quant.cpp) | 通过[aclnnAscendQuant](docs/aclnnAscendQuant.md)接口方式调用AscendQuantV2算子。 | | 133 | | aclnn接口 | [test_aclnn_ascend_quant](examples/test_aclnn_ascend_quant.cpp) | 通过[aclnnAscendQuant](docs/aclnnAscendQuant.md)接口方式调用AscendQuantV2算子。 | |
| 133 | | aclnn接口 | [test_aclnn_ascend_quant_v3](examples/test_aclnn_ascend_quant_v3.cpp) | 通过[aclnnAscendQuantV3](docs/aclnnAscendQuantV3.md)接口方式调用AscendQuantV2算子。 | | 134 | | aclnn接口 | [test_aclnn_ascend_quant_v3](examples/test_aclnn_ascend_quant_v3.cpp) | 通过[aclnnAscendQuantV3](docs/aclnnAscendQuantV3.md)接口方式调用AscendQuantV2算子。 | |
| 134 | -| 图模式 | - | 通过[算子IR](op_graph/ascend_quant_v2_proto.h)构图方式调用AscendQuantV2算子。 | | 135 | +| 图模式 | - | 通过[算子IR](op_graph/ascend_quant_v2_proto.h)构图方式调用AscendQuantV2算子。 | |
| @@ -183,7 +183,6 @@ aclnnStatus aclnnAscendQuant( | |||
| 183 | - 出参`y`数据类型仅支持INT8,数据格式不支持NZ。 | 183 | - 出参`y`数据类型仅支持INT8,数据格式不支持NZ。 |
| 184 | - 入参`dstType`仅支持取值2,表示INT8。 | 184 | - 入参`dstType`仅支持取值2,表示INT8。 |
| 185 | 185 | ||
| 186 | - | ||
| 187 | - **返回值:** | 186 | - **返回值:** |
| 188 | 187 | ||
| 189 | aclnnStatus:返回状态码,具体参见[aclnn返回码](../../../docs/zh/context/aclnn返回码.md)。 | 188 | aclnnStatus:返回状态码,具体参见[aclnn返回码](../../../docs/zh/context/aclnn返回码.md)。 |
| @@ -217,7 +216,8 @@ aclnnStatus aclnnAscendQuant( | |||
| 217 | <td>x、scale、offset、y的shape不满足限制条件。</td> | 216 | <td>x、scale、offset、y的shape不满足限制条件。</td> |
| 218 | </tr> | 217 | </tr> |
| 219 | <tr> | 218 | <tr> |
| 220 | - <td>roundMode不在有效取值范围。</tr> | 219 | + <td>roundMode不在有效取值范围。</td> |
| 220 | + </tr> | ||
| 221 | <tr> | 221 | <tr> |
| 222 | <td>dstType不在有效取值范围。</td> | 222 | <td>dstType不在有效取值范围。</td> |
| 223 | </tr> | 223 | </tr> |
| @@ -59,7 +59,6 @@ aclnnStatus aclnnAscendQuantV3( | |||
| 59 | 59 | ||
| 60 | - **参数说明:** | 60 | - **参数说明:** |
| 61 | 61 | ||
| 62 | - | ||
| 63 | <table style="undefined;table-layout: fixed; width: 1550px"><colgroup> | 62 | <table style="undefined;table-layout: fixed; width: 1550px"><colgroup> |
| 64 | <col style="width: 170px"> | 63 | <col style="width: 170px"> |
| 65 | <col style="width: 120px"> | 64 | <col style="width: 120px"> |
| @@ -106,7 +105,7 @@ aclnnStatus aclnnAscendQuantV3( | |||
| 106 | <td>offset(aclTensor*)</td> | 105 | <td>offset(aclTensor*)</td> |
| 107 | <td>输入</td> | 106 | <td>输入</td> |
| 108 | <td>可选参数,反量化中的offset值。对应公式中的`offset`。</td> | 107 | <td>可选参数,反量化中的offset值。对应公式中的`offset`。</td> |
| 109 | - <td><ul><li>支持空Tensor。</li><li>数据类型和shape需要与`scale`保持一致。</li><li>数据格式为NZ时,值为空,offset的数据类型和x保持一致。</li></td> | 108 | + <td><ul><li>支持空Tensor。</li><li>数据类型和shape需要与`scale`保持一致。</li><li>数据格式为NZ时,值为空,offset的数据类型和x保持一致。</li></ul></td> |
| 110 | <td>FLOAT32、FLOAT16、BFLOAT16</td> | 109 | <td>FLOAT32、FLOAT16、BFLOAT16</td> |
| 111 | <td>ND、NZ</td> | 110 | <td>ND、NZ</td> |
| 112 | <td>1-8</td> | 111 | <td>1-8</td> |
| @@ -156,7 +155,7 @@ aclnnStatus aclnnAscendQuantV3( | |||
| 156 | <td>y(aclTensor*)</td> | 155 | <td>y(aclTensor*)</td> |
| 157 | <td>输出</td> | 156 | <td>输出</td> |
| 158 | <td>量化的计算输出。对应公式中的`y`。</td> | 157 | <td>量化的计算输出。对应公式中的`y`。</td> |
| 159 | - <td><ul><li>支持空Tensor。</li><li>类型为INT32时,shape的最后一维是`x`最后一维的1/8,其余维度和`x`一致;其他类型时,shape与`x`一致。</li></td> | 158 | + <td><ul><li>支持空Tensor。</li><li>类型为INT32时,shape的最后一维是`x`最后一维的1/8,其余维度和`x`一致;其他类型时,shape与`x`一致。</li></ul></td> |
| 160 | <td>INT8、INT32、INT4、HIFLOAT8、FLOAT8_E5M2、FLOAT8_E4M3FN</td> | 159 | <td>INT8、INT32、INT4、HIFLOAT8、FLOAT8_E5M2、FLOAT8_E4M3FN</td> |
| 161 | <td>ND、NZ</td> | 160 | <td>ND、NZ</td> |
| 162 | <td>1-8</td> | 161 | <td>1-8</td> |
| @@ -193,13 +192,11 @@ aclnnStatus aclnnAscendQuantV3( | |||
| 193 | - 入参`dstType`支持取值2,3,29,分别表示INT8、INT32、INT4。当输入`x`的数据格式为NZ时,支持取值3,表示INT32。 | 192 | - 入参`dstType`支持取值2,3,29,分别表示INT8、INT32、INT4。当输入`x`的数据格式为NZ时,支持取值3,表示INT32。 |
| 194 | - 入参`axis`支持指定x的最后两个维度(假设输入x维度是xDimNum,axis取值范围是[-2,-1]或[xDimNum-2,xDimNum-1])。 | 193 | - 入参`axis`支持指定x的最后两个维度(假设输入x维度是xDimNum,axis取值范围是[-2,-1]或[xDimNum-2,xDimNum-1])。 |
| 195 | 194 | ||
| 196 | - | ||
| 197 | - <term>Ascend 950PR/Ascend 950DT</term>: | 195 | - <term>Ascend 950PR/Ascend 950DT</term>: |
| 198 | - 参数`x`、`scale`、`offset`的数据格式不支持NZ。 | 196 | - 参数`x`、`scale`、`offset`的数据格式不支持NZ。 |
| 199 | - 入参`roundMode`:`dstType`表示FLOAT8_E5M2或FLOAT8_E4M3FN时,只支持round。`dstType`表示HIFLOAT8时,支持round和hybrid。`dstType`表示其他类型时,支持round,ceil,trunc和floor。 | 197 | - 入参`roundMode`:`dstType`表示FLOAT8_E5M2或FLOAT8_E4M3FN时,只支持round。`dstType`表示HIFLOAT8时,支持round和hybrid。`dstType`表示其他类型时,支持round,ceil,trunc和floor。 |
| 200 | - 入参`axis`支持指定x的最后两个维度(假设输入x维度是xDimNum,axis取值范围是[-2,-1]或[xDimNum-2,xDimNum-1])。 | 198 | - 入参`axis`支持指定x的最后两个维度(假设输入x维度是xDimNum,axis取值范围是[-2,-1]或[xDimNum-2,xDimNum-1])。 |
| 201 | 199 | ||
| 202 | - | ||
| 203 | - <term>Atlas 推理系列产品</term>: | 200 | - <term>Atlas 推理系列产品</term>: |
| 204 | - 入参`x`、`scale`、`offset`的数据类型不支持BFLOAT16,数据格式不支持NZ。 | 201 | - 入参`x`、`scale`、`offset`的数据类型不支持BFLOAT16,数据格式不支持NZ。 |
| 205 | - 出参`y`数据类型仅支持INT8,数据格式不支持NZ。 | 202 | - 出参`y`数据类型仅支持INT8,数据格式不支持NZ。 |
| @@ -207,7 +204,6 @@ aclnnStatus aclnnAscendQuantV3( | |||
| 207 | - 入参`dstType`仅支持取值2,表示INT8。 | 204 | - 入参`dstType`仅支持取值2,表示INT8。 |
| 208 | - 入参`axis`只支持指定x的最后一个维度(假设输入x维度是xDimNum,axis取值是-1或xDimNum-1)。 | 205 | - 入参`axis`只支持指定x的最后一个维度(假设输入x维度是xDimNum,axis取值是-1或xDimNum-1)。 |
| 209 | 206 | ||
| 210 | - | ||
| 211 | - **返回值:** | 207 | - **返回值:** |
| 212 | 208 | ||
| 213 | aclnnStatus:返回状态码,具体参见[aclnn返回码](../../../docs/zh/context/aclnn返回码.md)。 | 209 | aclnnStatus:返回状态码,具体参见[aclnn返回码](../../../docs/zh/context/aclnn返回码.md)。 |
| @@ -244,7 +240,8 @@ aclnnStatus aclnnAscendQuantV3( | |||
| 244 | <td>x的维数不在1到8维之间。</td> | 240 | <td>x的维数不在1到8维之间。</td> |
| 245 | </tr> | 241 | </tr> |
| 246 | <tr> | 242 | <tr> |
| 247 | - <td>roundMode不在有效取值范围。</tr> | 243 | + <td>roundMode不在有效取值范围。</td> |
| 244 | + </tr> | ||
| 248 | <tr> | 245 | <tr> |
| 249 | <td>dstType不在有效取值范围。</td> | 246 | <td>dstType不在有效取值范围。</td> |
| 250 | </tr> | 247 | </tr> |
| @@ -93,6 +93,7 @@ | |||
| 93 | </tbody></table> | 93 | </tbody></table> |
| 94 | 94 | ||
| 95 | - Kirin X90/Kirin 9030 处理器系列产品: 不支持BFLOAT16。 | 95 | - Kirin X90/Kirin 9030 处理器系列产品: 不支持BFLOAT16。 |
| 96 | + | ||
| 96 | ## 约束说明 | 97 | ## 约束说明 |
| 97 | 98 | ||
| 98 | 输入和输出参数中shape的N和M必须是正整数,且M的取值小于等于25000。 | 99 | 输入和输出参数中shape的N和M必须是正整数,且M的取值小于等于25000。 |
| @@ -60,7 +60,6 @@ | |||
| 60 | 60 | ||
| 61 | 其中,x\_glu表示dequantOut<sub>i</sub>的偶数索引部分,x\_linear表示dequantOut<sub>i</sub>的奇数索引部分。 | 61 | 其中,x\_glu表示dequantOut<sub>i</sub>的偶数索引部分,x\_linear表示dequantOut<sub>i</sub>的奇数索引部分。 |
| 62 | 62 | ||
| 63 | - | ||
| 64 | ## 参数说明 | 63 | ## 参数说明 |
| 65 | 64 | ||
| 66 | <table style="undefined;table-layout: fixed; width: 951px"><colgroup> | 65 | <table style="undefined;table-layout: fixed; width: 951px"><colgroup> |
| @@ -207,13 +206,12 @@ | |||
| 207 | </tr> | 206 | </tr> |
| 208 | </tbody></table> | 207 | </tbody></table> |
| 209 | 208 | ||
| 209 | +- Kirin X90/Kirin 9030 处理器系列产品: | ||
| 210 | + - 输入`x`:数据类型不支持BFLOAT16。 | ||
| 211 | + - 输入`biasOptional`:数据类型不支持BFLOAT16。 | ||
| 212 | + - 输入`quantScaleOptional`:数据类型不支持FLOAT16。 | ||
| 213 | + - 输出`y`:数据类型仅支持INT8。 | ||
| 210 | 214 | ||
| 211 | - | ||
| 212 | -- Kirin X90/Kirin 9030 处理器系列产品: | ||
| 213 | - - 输入`x`:数据类型不支持BFLOAT16。 | ||
| 214 | - - 输入`biasOptional`:数据类型不支持BFLOAT16。 | ||
| 215 | - - 输入`quantScaleOptional`:数据类型不支持FLOAT16。 | ||
| 216 | - - 输出`y`:数据类型仅支持INT8 | ||
| 217 | ## 约束说明 | 215 | ## 约束说明 |
| 218 | 216 | ||
| 219 | - <term>Ascend 950PR/Ascend 950DT</term>: | 217 | - <term>Ascend 950PR/Ascend 950DT</term>: |
| @@ -230,7 +228,6 @@ | |||
| 230 | - 当quant_mode为static时,quant_scale和quant_offset为1维,值为1;quant_mode为dynamic时,quant_scale和quant_offset | 228 | - 当quant_mode为static时,quant_scale和quant_offset为1维,值为1;quant_mode为dynamic时,quant_scale和quant_offset |
| 231 | - 算子支持的输入张量的内存大小有上限,校验公式:weight_scale张量内存大小+bias张量内存大小+quant_scale张量内存大小+quant_offset张量内存大小 + (activation_scale张量内存大小 + scale张量内存大小)/40 + x张量最后一维H内存大小 * 10 < 192KB。 | 229 | - 算子支持的输入张量的内存大小有上限,校验公式:weight_scale张量内存大小+bias张量内存大小+quant_scale张量内存大小+quant_offset张量内存大小 + (activation_scale张量内存大小 + scale张量内存大小)/40 + x张量最后一维H内存大小 * 10 < 192KB。 |
| 232 | 230 | ||
| 233 | - | ||
| 234 | ## 调用说明 | 231 | ## 调用说明 |
| 235 | 232 | ||
| 236 | | 调用方式 | 调用样例 | 说明 | | 233 | | 调用方式 | 调用样例 | 说明 | |
| @@ -107,6 +107,7 @@ aclnnStatus aclnnDequantSwigluQuant( | |||
| 107 | <td>1或2</td> | 107 | <td>1或2</td> |
| 108 | <td>x</td> | 108 | <td>x</td> |
| 109 | </tr> | 109 | </tr> |
| 110 | + <tr> | ||
| 110 | <td>activationScaleOptional(aclTensor*)</td> | 111 | <td>activationScaleOptional(aclTensor*)</td> |
| 111 | <td>输入</td> | 112 | <td>输入</td> |
| 112 | <td>激活函数的反量化scale,公式中的activationScaleOptional。</td> | 113 | <td>激活函数的反量化scale,公式中的activationScaleOptional。</td> |
| @@ -140,6 +140,7 @@ aclnnStatus aclnnDequantSwigluQuantV2( | |||
| 140 | <td>1或2</td> | 140 | <td>1或2</td> |
| 141 | <td>x</td> | 141 | <td>x</td> |
| 142 | </tr> | 142 | </tr> |
| 143 | + <tr> | ||
| 143 | <td>activationScaleOptional(aclTensor*)</td> | 144 | <td>activationScaleOptional(aclTensor*)</td> |
| 144 | <td>输入</td> | 145 | <td>输入</td> |
| 145 | <td>激活函数的反量化scale。</td> | 146 | <td>激活函数的反量化scale。</td> |
| @@ -173,7 +174,7 @@ aclnnStatus aclnnDequantSwigluQuantV2( | |||
| 173 | <td>quantOffsetOptional(aclTensor*)</td> | 174 | <td>quantOffsetOptional(aclTensor*)</td> |
| 174 | <td>输入</td> | 175 | <td>输入</td> |
| 175 | <td>量化的offset。</td> | 176 | <td>量化的offset。</td> |
| 176 | - <td><ul><li>quant_mode为动态时不需要quantOffset输入,静态量化中quantOffset必须输入,且数据类型与shape同quantScale。</td> | 177 | + <td><ul><li>quant_mode为动态时不需要quantOffset输入,静态量化中quantOffset必须输入,且数据类型与shape同quantScale。</li></ul></td> |
| 177 | <td>FLOAT</td> | 178 | <td>FLOAT</td> |
| 178 | <td>ND</td> | 179 | <td>ND</td> |
| 179 | <td>-</td> | 180 | <td>-</td> |
| @@ -193,7 +194,7 @@ aclnnStatus aclnnDequantSwigluQuantV2( | |||
| 193 | <td>activateLeft(bool)</td> | 194 | <td>activateLeft(bool)</td> |
| 194 | <td>输入</td> | 195 | <td>输入</td> |
| 195 | <td>表示是否对输入的左半部分做swiglu激活。</td> | 196 | <td>表示是否对输入的左半部分做swiglu激活。</td> |
| 196 | - <td><ul><li>当值为false时,对输入的右半部分做激活。如果swigluMode为1,activateLeft必须为true。</td> | 197 | + <td><ul><li>当值为false时,对输入的右半部分做激活。如果swigluMode为1,activateLeft必须为true。</li></ul></td> |
| 197 | <td>-</td> | 198 | <td>-</td> |
| 198 | <td>-</td> | 199 | <td>-</td> |
| 199 | <td>-</td> | 200 | <td>-</td> |
| @@ -590,4 +590,4 @@ aclnnStatus aclnnDynamicBlockQuant( | |||
| 590 | 590 | ||
| 591 | return 0; | 591 | return 0; |
| 592 | } | 592 | } |
| 593 | - ``` | 593 | + ``` |
| @@ -109,7 +109,7 @@ aclnnStatus aclnnDynamicDualLevelMxQuant( | |||
| 109 | <td>x (aclTensor*)</td> | 109 | <td>x (aclTensor*)</td> |
| 110 | <td>输入</td> | 110 | <td>输入</td> |
| 111 | <td>表示输入x,对应公式中<em>x</em><sub>i</sub>。</td> | 111 | <td>表示输入x,对应公式中<em>x</em><sub>i</sub>。</td> |
| 112 | - <td><ul><li>x的最后一维必须是偶数;</li><li> 不支持空Tensor。</td> | 112 | + <td><ul><li>x的最后一维必须是偶数;</li><li> 不支持空Tensor。</li></ul></td> |
| 113 | <td>FLOAT16、BFLOAT16</td> | 113 | <td>FLOAT16、BFLOAT16</td> |
| 114 | <td>ND</td> | 114 | <td>ND</td> |
| 115 | <td>1-7</td> | 115 | <td>1-7</td> |
| @@ -208,7 +208,6 @@ aclnnStatus aclnnDynamicDualLevelMxQuant( | |||
| 208 | </tbody> | 208 | </tbody> |
| 209 | </table> | 209 | </table> |
| 210 | 210 | ||
| 211 | - | ||
| 212 | - **返回值:** | 211 | - **返回值:** |
| 213 | 212 | ||
| 214 | aclnnStatus:返回状态码,具体参见[aclnn返回码](../../../docs/zh/context/aclnn返回码.md)。 | 213 | aclnnStatus:返回状态码,具体参见[aclnn返回码](../../../docs/zh/context/aclnn返回码.md)。 |
| @@ -303,7 +302,6 @@ aclnnStatus aclnnDynamicDualLevelMxQuant( | |||
| 303 | - 其他维度与输入x一致。 | 302 | - 其他维度与输入x一致。 |
| 304 | - 确定性说明:aclnnDynamicDualLevelMxQuant默认确定性实现。 | 303 | - 确定性说明:aclnnDynamicDualLevelMxQuant默认确定性实现。 |
| 305 | 304 | ||
| 306 | - | ||
| 307 | ## 调用示例 | 305 | ## 调用示例 |
| 308 | 306 | ||
| 309 | 示例代码如下,仅供参考,具体编译和执行过程请参考[编译与运行样例](../../../docs/zh/context/编译与运行样例.md)。 | 307 | 示例代码如下,仅供参考,具体编译和执行过程请参考[编译与运行样例](../../../docs/zh/context/编译与运行样例.md)。 |
| @@ -520,4 +518,4 @@ aclnnStatus aclnnDynamicDualLevelMxQuant( | |||
| 520 | Finalize(deviceId, stream); | 518 | Finalize(deviceId, stream); |
| 521 | return 0; | 519 | return 0; |
| 522 | } | 520 | } |
| 523 | - ``` | 521 | +``` |
| @@ -1,6 +1,6 @@ | |||
| 1 | # DynamicMxQuant | 1 | # DynamicMxQuant |
| 2 | 2 | ||
| 3 | -## 产品支持情况 | 3 | +## 产品支持情况 |
| 4 | 4 | ||
| 5 | | 产品 | 是否支持 | | 5 | | 产品 | 是否支持 | |
| 6 | | :----------------------------------------------------------- | :------: | | 6 | | :----------------------------------------------------------- | :------: | |
| @@ -75,14 +75,14 @@ | |||
| 75 | <td>INT64</td> | 75 | <td>INT64</td> |
| 76 | <td>ND</td> | 76 | <td>ND</td> |
| 77 | </tr> | 77 | </tr> |
| 78 | - </tr> | 78 | + <tr> |
| 79 | <td>y</td> | 79 | <td>y</td> |
| 80 | <td>输出</td> | 80 | <td>输出</td> |
| 81 | <td>输入x量化后的对应结果</td> | 81 | <td>输入x量化后的对应结果</td> |
| 82 | <td>FLOAT4_E2M1、FLOAT4_E1M2、FLOAT8_E4M3FN、FLOAT8_E5M2</td> | 82 | <td>FLOAT4_E2M1、FLOAT4_E1M2、FLOAT8_E4M3FN、FLOAT8_E5M2</td> |
| 83 | <td>ND</td> | 83 | <td>ND</td> |
| 84 | </tr> | 84 | </tr> |
| 85 | - </tr> | 85 | + <tr> |
| 86 | <td>mxscale</td> | 86 | <td>mxscale</td> |
| 87 | <td>输出</td> | 87 | <td>输出</td> |
| 88 | <td>每个分组对应的量化尺度</td> | 88 | <td>每个分组对应的量化尺度</td> |
| @@ -104,4 +104,4 @@ | |||
| 104 | 104 | ||
| 105 | | 调用方式 | 调用样例 | 说明 | | 105 | | 调用方式 | 调用样例 | 说明 | |
| 106 | |--------------|------------------------------------------------------------------------|--------------------------------------------------------------| | 106 | |--------------|------------------------------------------------------------------------|--------------------------------------------------------------| |
| 107 | -| aclnn调用 | [test_aclnn_dynamic_mx_quant](./examples/test_aclnn_dynamic_mx_quant.cpp) | 通过[aclnnDynamicMxQuant](./docs/aclnnDynamicMxQuant.md)接口方式调用DynamicMxQuant算子。 | | 107 | +| aclnn调用 | [test_aclnn_dynamic_mx_quant](./examples/test_aclnn_dynamic_mx_quant.cpp) | 通过[aclnnDynamicMxQuant](./docs/aclnnDynamicMxQuant.md)接口方式调用DynamicMxQuant算子。 | |
| @@ -37,6 +37,7 @@ | |||
| 37 | | FLOAT4_E1M2 | 0 | | 37 | | FLOAT4_E1M2 | 0 | |
| 38 | | FLOAT8_E4M3FN | 8 | | 38 | | FLOAT8_E4M3FN | 8 | |
| 39 | | FLOAT8_E5M2 | 15 | | 39 | | FLOAT8_E5M2 | 15 | |
| 40 | + | ||
| 40 | - 场景2,当scaleAlg为1时,只涉及FP8类型: | 41 | - 场景2,当scaleAlg为1时,只涉及FP8类型: |
| 41 | - 将长向量按块分,每块长度为k,对每块单独计算一个块缩放因子$S_{fp32}^b$,再把块内所有元素用同一个$S_{fp32}^b$映射到目标低精度类型FP8。如果最后一块不足k个元素,把缺失值视为0,按照完整块处理。 | 42 | - 将长向量按块分,每块长度为k,对每块单独计算一个块缩放因子$S_{fp32}^b$,再把块内所有元素用同一个$S_{fp32}^b$映射到目标低精度类型FP8。如果最后一块不足k个元素,把缺失值视为0,按照完整块处理。 |
| 42 | - 找到该块中数值的最大绝对值: | 43 | - 找到该块中数值的最大绝对值: |
| @@ -308,7 +309,6 @@ aclnnStatus aclnnDynamicMxQuant( | |||
| 308 | - mxscaleOut.shape[-1] = 2。 | 309 | - mxscaleOut.shape[-1] = 2。 |
| 309 | - 其他维度与输入x一致。 | 310 | - 其他维度与输入x一致。 |
| 310 | 311 | ||
| 311 | - | ||
| 312 | ## 调用示例 | 312 | ## 调用示例 |
| 313 | 313 | ||
| 314 | 示例代码如下,仅供参考,具体编译和执行过程请参考[编译与运行样例](../../../docs/zh/context/编译与运行样例.md)。 | 314 | 示例代码如下,仅供参考,具体编译和执行过程请参考[编译与运行样例](../../../docs/zh/context/编译与运行样例.md)。 |
| @@ -496,4 +496,4 @@ int main() | |||
| 496 | Finalize(deviceId, stream); | 496 | Finalize(deviceId, stream); |
| 497 | return 0; | 497 | return 0; |
| 498 | } | 498 | } |
| 499 | -``` | 499 | +``` |
| @@ -71,7 +71,7 @@ | |||
| 71 | 71 | ||
| 72 | ## 约束说明 | 72 | ## 约束说明 |
| 73 | 73 | ||
| 74 | - - 关于x、mxscale1、mxscale2的shape约束说明如下: | 74 | + - 关于x、mxscale1、mxscale2的shape约束说明如下: |
| 75 | - x的维度应该大于等于2。 | 75 | - x的维度应该大于等于2。 |
| 76 | - rank(mxscale1) = rank(x) + 1。 | 76 | - rank(mxscale1) = rank(x) + 1。 |
| 77 | - rank(mxscale2) = rank(x) + 1。 | 77 | - rank(mxscale2) = rank(x) + 1。 |
| @@ -87,4 +87,4 @@ | |||
| 87 | | 调用方式 | 样例代码 | 说明 | | 87 | | 调用方式 | 样例代码 | 说明 | |
| 88 | | ---------------- | --------------------------- | --------------------------------------------------- | | 88 | | ---------------- | --------------------------- | --------------------------------------------------- | |
| 89 | | aclnn接口 | [test_aclnn_dynamic_mx_quant_with_dual_axis](examples/arch35/test_aclnn_dynamic_mx_quant_with_dual_axis.cpp) | 通过[aclnnDynamicMxQuantWithDualAxis](docs/aclnnDynamicMxQuantWithDualAxis.md)接口方式调用DynamicMxQuantWithDualAxis算子。 | | 89 | | aclnn接口 | [test_aclnn_dynamic_mx_quant_with_dual_axis](examples/arch35/test_aclnn_dynamic_mx_quant_with_dual_axis.cpp) | 通过[aclnnDynamicMxQuantWithDualAxis](docs/aclnnDynamicMxQuantWithDualAxis.md)接口方式调用DynamicMxQuantWithDualAxis算子。 | |
| 90 | -| 图模式 | - | 通过[算子IR](op_graph/dynamic_mx_quant_with_dual_axis_proto.h)构图方式调用DynamicMxQuantWithDualAxis算子。 | | 90 | +| 图模式 | - | 通过[算子IR](op_graph/dynamic_mx_quant_with_dual_axis_proto.h)构图方式调用DynamicMxQuantWithDualAxis算子。 | |
| @@ -113,9 +113,10 @@ | |||
| 113 | - <term>Atlas A3 训练系列产品/Atlas A3 推理系列产品</term>、<term>Atlas A2 训练系列产品/Atlas A2 推理系列产品</term>:输出`y`的数据类型仅支持INT8、INT4。 | 113 | - <term>Atlas A3 训练系列产品/Atlas A3 推理系列产品</term>、<term>Atlas A2 训练系列产品/Atlas A2 推理系列产品</term>:输出`y`的数据类型仅支持INT8、INT4。 |
| 114 | 114 | ||
| 115 | - Kirin X90/Kirin 9030 处理器系列产品: | 115 | - Kirin X90/Kirin 9030 处理器系列产品: |
| 116 | - - 输入`x`:数据类型仅支持FLOAT16。 | 116 | + - 输入`x`:数据类型仅支持FLOAT16。 |
| 117 | - - 可选输入`smooth_scales`:数据类型仅支持FLOAT16。 | 117 | + - 可选输入`smooth_scales`:数据类型仅支持FLOAT16。 |
| 118 | - - 输出`y`:数据类型仅支持INT8。 | 118 | + - 输出`y`:数据类型仅支持INT8。 |
| 119 | + | ||
| 119 | ## 约束说明 | 120 | ## 约束说明 |
| 120 | 121 | ||
| 121 | 无 | 122 | 无 |
| @@ -125,4 +126,4 @@ | |||
| 125 | | 调用方式 | 样例代码 | 说明 | | 126 | | 调用方式 | 样例代码 | 说明 | |
| 126 | | ---------------- | --------------------------- | --------------------------------------------------- | | 127 | | ---------------- | --------------------------- | --------------------------------------------------- | |
| 127 | | aclnn接口 | [test_aclnn_dynamic_quant](examples/test_aclnn_dynamic_quant.cpp) | 通过[aclnnDynamicQuant](docs/aclnnDynamicQuant.md)接口方式调用DynamicQuant算子。 | | 128 | | aclnn接口 | [test_aclnn_dynamic_quant](examples/test_aclnn_dynamic_quant.cpp) | 通过[aclnnDynamicQuant](docs/aclnnDynamicQuant.md)接口方式调用DynamicQuant算子。 | |
| 128 | -| 图模式 | - | 通过[算子IR](op_graph/dynamic_quant_proto.h)构图方式调用DynamicQuant算子。 | | 129 | +| 图模式 | - | 通过[算子IR](op_graph/dynamic_quant_proto.h)构图方式调用DynamicQuant算子。 | |
| @@ -266,7 +266,7 @@ aclnnStatus aclnnDynamicQuantV3( | |||
| 266 | 266 | ||
| 267 | ## aclnnDynamicQuantV3 | 267 | ## aclnnDynamicQuantV3 |
| 268 | 268 | ||
| 269 | - - **参数说明:** | 269 | +- **参数说明:** |
| 270 | 270 | ||
| 271 | <table style="undefined;table-layout: fixed; width: 953px"><colgroup> | 271 | <table style="undefined;table-layout: fixed; width: 953px"><colgroup> |
| 272 | <col style="width: 173px"> | 272 | <col style="width: 173px"> |
| @@ -312,9 +312,9 @@ aclnnStatus aclnnDynamicQuantV3( | |||
| 312 | - 确定性计算: | 312 | - 确定性计算: |
| 313 | - aclnnDynamicQuantV3默认确定性实现。 | 313 | - aclnnDynamicQuantV3默认确定性实现。 |
| 314 | 314 | ||
| 315 | -yOut的数据类型为INT4时,需满足x和yOut的最后一维能被2整除。 | 315 | + yOut的数据类型为INT4时,需满足x和yOut的最后一维能被2整除。 |
| 316 | -yOut的数据类型为INT32时,需满足x的最后一维能被8整除。 | 316 | + yOut的数据类型为INT32时,需满足x的最后一维能被8整除。 |
| 317 | -当有groupIndexOptional时,专家数不超过x剔除最后一维的各个维度乘积。groupIndexOptional的值需要是一组不小于零且非递减的数组,且最后一个值和x剔除最后一维的各个维度乘积相等。若不满足该条件,结果无实际意义。 | 317 | + 当有groupIndexOptional时,专家数不超过x剔除最后一维的各个维度乘积。groupIndexOptional的值需要是一组不小于零且非递减的数组,且最后一个值和x剔除最后一维的各个维度乘积相等。若不满足该条件,结果无实际意义。 |
| 318 | 318 | ||
| 319 | ## 调用示例 | 319 | ## 调用示例 |
| 320 | 320 | ||
| @@ -1,6 +1,6 @@ | |||
| 1 | # DynamicQuantUpdateScatter | 1 | # DynamicQuantUpdateScatter |
| 2 | 2 | ||
| 3 | -## 产品支持情况 | 3 | +## 产品支持情况 |
| 4 | 4 | ||
| 5 | | 产品 | 是否支持 | | 5 | | 产品 | 是否支持 | |
| 6 | | ---- | :----:| | 6 | | ---- | :----:| |
| @@ -82,9 +82,10 @@ | |||
| 82 | </tr> | 82 | </tr> |
| 83 | </tbody></table> | 83 | </tbody></table> |
| 84 | 84 | ||
| 85 | -- Kirin X90/Kirin 9030 处理器系列产品: | 85 | +- Kirin X90/Kirin 9030 处理器系列产品: |
| 86 | - - 输入`updates`:数据类型仅支持FLOAT16。 | 86 | + - 输入`updates`:数据类型仅支持FLOAT16。 |
| 87 | - - 输入`smooth_scales`:数据类型仅支持FLOAT16。 | 87 | + - 输入`smooth_scales`:数据类型仅支持FLOAT16。 |
| 88 | + | ||
| 88 | ## 约束说明 | 89 | ## 约束说明 |
| 89 | 90 | ||
| 90 | 1. indices的维数只能是1维或者2维,如果是2维,其第2维的大小必须是2。 | 91 | 1. indices的维数只能是1维或者2维,如果是2维,其第2维的大小必须是2。 |
| @@ -1,6 +1,6 @@ | |||
| 1 | # DynamicQuantUpdateScatterV2 | 1 | # DynamicQuantUpdateScatterV2 |
| 2 | 2 | ||
| 3 | -## 产品支持情况 | 3 | +## 产品支持情况 |
| 4 | 4 | ||
| 5 | | 产品 | 是否支持 | | 5 | | 产品 | 是否支持 | |
| 6 | | ---- | :----:| | 6 | | ---- | :----:| |
| @@ -69,8 +69,9 @@ | |||
| 69 | </tr> | 69 | </tr> |
| 70 | </tbody></table> | 70 | </tbody></table> |
| 71 | 71 | ||
| 72 | -- Kirin X90/Kirin 9030 处理器系列产品: | 72 | +- Kirin X90/Kirin 9030 处理器系列产品: |
| 73 | - - 输入`x`:数据类型仅支持FLOAT16。 | 73 | + - 输入`x`:数据类型仅支持FLOAT16。 |
| 74 | + | ||
| 74 | ## 约束说明 | 75 | ## 约束说明 |
| 75 | 76 | ||
| 76 | - 量化方式支持非对称量化,量化数据类型支持INT4。 | 77 | - 量化方式支持非对称量化,量化数据类型支持INT4。 |
| @@ -172,10 +172,11 @@ | |||
| 172 | - 输入`dst_type`:只支持配置为2。 | 172 | - 输入`dst_type`:只支持配置为2。 |
| 173 | - <term>Atlas A3 训练系列产品/Atlas A3 推理系列产品</term>、<term>Atlas A2 训练系列产品/Atlas A2 推理系列产品</term>:输出`y`的数据类型仅支持INT8、INT4。 | 173 | - <term>Atlas A3 训练系列产品/Atlas A3 推理系列产品</term>、<term>Atlas A2 训练系列产品/Atlas A2 推理系列产品</term>:输出`y`的数据类型仅支持INT8、INT4。 |
| 174 | 174 | ||
| 175 | -- Kirin X90/Kirin 9030 处理器系列产品: | 175 | +- Kirin X90/Kirin 9030 处理器系列产品: |
| 176 | - - 输入`x`:数据类型仅支持FLOAT16。 | 176 | + - 输入`x`:数据类型仅支持FLOAT16。 |
| 177 | - - 可选输入`smooth_scales`:数据类型仅支持FLOAT16。 | 177 | + - 可选输入`smooth_scales`:数据类型仅支持FLOAT16。 |
| 178 | - - 输出`y`:数据类型仅支持INT8。 | 178 | + - 输出`y`:数据类型仅支持INT8。 |
| 179 | + | ||
| 179 | ## 约束说明 | 180 | ## 约束说明 |
| 180 | 181 | ||
| 181 | - E不应大于x去掉最后一个维度后的维度的乘积结果(S)。 | 182 | - E不应大于x去掉最后一个维度后的维度的乘积结果(S)。 |
| @@ -186,4 +187,4 @@ | |||
| 186 | | 调用方式 | 样例代码 | 说明 | | 187 | | 调用方式 | 样例代码 | 说明 | |
| 187 | | ---------------- | --------------------------- | --------------------------------------------------- | | 188 | | ---------------- | --------------------------- | --------------------------------------------------- | |
| 188 | | aclnn接口 | [test_aclnn_dynamic_quant_v2](examples/test_aclnn_dynamic_quant_v2.cpp) | 通过[aclnnDynamicQuantV2](docs/aclnnDynamicQuantV2.md)接口方式调用DynamicQuantV2算子。 | | 189 | | aclnn接口 | [test_aclnn_dynamic_quant_v2](examples/test_aclnn_dynamic_quant_v2.cpp) | 通过[aclnnDynamicQuantV2](docs/aclnnDynamicQuantV2.md)接口方式调用DynamicQuantV2算子。 | |
| 189 | -| 图模式 | - | 通过[算子IR](op_graph/dynamic_quant_v2_proto.h)构图方式调用DynamicQuantV2算子。 | | 190 | +| 图模式 | - | 通过[算子IR](op_graph/dynamic_quant_v2_proto.h)构图方式调用DynamicQuantV2算子。 | |
| @@ -1,6 +1,6 @@ | |||
| 1 | # FakeQuantAffineCachemask | 1 | # FakeQuantAffineCachemask |
| 2 | 2 | ||
| 3 | -## 产品支持情况 | 3 | +## 产品支持情况 |
| 4 | 4 | ||
| 5 | | 产品 | 是否支持 | | 5 | | 产品 | 是否支持 | |
| 6 | | ---- | :----:| | 6 | | ---- | :----:| |
| @@ -107,6 +107,7 @@ | |||
| 107 | </tbody></table> | 107 | </tbody></table> |
| 108 | 108 | ||
| 109 | - Kirin X90/Kirin 9030 处理器系列产品: `zero_point`支持INT32、FLOAT、FLOAT16。 | 109 | - Kirin X90/Kirin 9030 处理器系列产品: `zero_point`支持INT32、FLOAT、FLOAT16。 |
| 110 | + | ||
| 110 | ## 约束说明 | 111 | ## 约束说明 |
| 111 | 112 | ||
| 112 | 无 | 113 | 无 |
| @@ -117,4 +118,4 @@ | |||
| 117 | | ---------------- | --------------------------- | --------------------------------------------------- | | 118 | | ---------------- | --------------------------- | --------------------------------------------------- | |
| 118 | | aclnn接口 | [test_aclnn_fake_quant_per_channel_affine_cachemask.cpp](examples/test_aclnn_fake_quant_per_channel_affine_cachemask.cpp) | 通过[aclnnFakeQuantPerChannelAffineCachemask](docs/aclnnFakeQuantPerChannelAffineCachemask.md)接口方式调用FakeQuantAffineCachemask算子。 | | 119 | | aclnn接口 | [test_aclnn_fake_quant_per_channel_affine_cachemask.cpp](examples/test_aclnn_fake_quant_per_channel_affine_cachemask.cpp) | 通过[aclnnFakeQuantPerChannelAffineCachemask](docs/aclnnFakeQuantPerChannelAffineCachemask.md)接口方式调用FakeQuantAffineCachemask算子。 | |
| 119 | | aclnn接口 | [test_aclnn_fake_quant_per_tensor_affine_cachemask.cpp](examples/test_aclnn_fake_quant_per_tensor_affine_cachemask.cpp) | 通过[aclnnFakeQuantPerTensorAffineCachemask](docs/aclnnFakeQuantPerTensorAffineCachemask.md)接口方式调用FakeQuantAffineCachemask算子。 | | 120 | | aclnn接口 | [test_aclnn_fake_quant_per_tensor_affine_cachemask.cpp](examples/test_aclnn_fake_quant_per_tensor_affine_cachemask.cpp) | 通过[aclnnFakeQuantPerTensorAffineCachemask](docs/aclnnFakeQuantPerTensorAffineCachemask.md)接口方式调用FakeQuantAffineCachemask算子。 | |
| 120 | -| 图模式 | [test_geir_fake_quant_affine_cachemask.cpp](examples/test_geir_fake_quant_affine_cachemask.cpp) | 通过[算子IR](op_graph/fake_quant_affine_cachemask_proto.h)构图方式调用FakeQuantAffineCachemask算子。 | | 121 | +| 图模式 | - | 通过[算子IR](op_graph/fake_quant_affine_cachemask_proto.h)构图方式调用FakeQuantAffineCachemask算子。 | |
| @@ -50,6 +50,7 @@ aclnnStatus aclnnFakeQuantPerChannelAffineCachemaskGetWorkspaceSize( | |||
| 50 | uint64_t *workspaceSize, | 50 | uint64_t *workspaceSize, |
| 51 | aclOpExecutor **executor) | 51 | aclOpExecutor **executor) |
| 52 | ``` | 52 | ``` |
| 53 | + | ||
| 53 | ```Cpp | 54 | ```Cpp |
| 54 | aclnnStatus aclnnFakeQuantPerChannelAffineCachemask( | 55 | aclnnStatus aclnnFakeQuantPerChannelAffineCachemask( |
| 55 | void *workspace, | 56 | void *workspace, |
| @@ -278,12 +279,12 @@ aclnnStatus aclnnFakeQuantPerChannelAffineCachemask( | |||
| 278 | aclnnStatus:返回状态码,具体参见[aclnn返回码](../../../docs/zh/context/aclnn返回码.md)。 | 279 | aclnnStatus:返回状态码,具体参见[aclnn返回码](../../../docs/zh/context/aclnn返回码.md)。 |
| 279 | 280 | ||
| 280 | ## 约束说明 | 281 | ## 约束说明 |
| 282 | + | ||
| 281 | - 确定性计算: | 283 | - 确定性计算: |
| 282 | - aclnnFakeQuantPerChannelAffineCachemask默认确定性实现。 | 284 | - aclnnFakeQuantPerChannelAffineCachemask默认确定性实现。 |
| 283 | 285 | ||
| 284 | - 当前新算子FakeQuantPerChannelAffineCachemask不支持zero_point的float32和float16输入,故先在aclnn接口内部拦截,待算子支持后放开该限制。 | 286 | - 当前新算子FakeQuantPerChannelAffineCachemask不支持zero_point的float32和float16输入,故先在aclnn接口内部拦截,待算子支持后放开该限制。 |
| 285 | 287 | ||
| 286 | - | ||
| 287 | ## 调用示例 | 288 | ## 调用示例 |
| 288 | 289 | ||
| 289 | 示例代码如下,仅供参考,具体编译和执行过程请参考[编译与运行样例](../../../docs/zh/context/编译与运行样例.md)。 | 290 | 示例代码如下,仅供参考,具体编译和执行过程请参考[编译与运行样例](../../../docs/zh/context/编译与运行样例.md)。 |
| @@ -444,4 +445,3 @@ int main() { | |||
| 444 | return 0; | 445 | return 0; |
| 445 | } | 446 | } |
| 446 | ``` | 447 | ``` |
| 447 | - | ||
| @@ -47,6 +47,7 @@ aclnnStatus aclnnFakeQuantPerTensorAffineCachemaskGetWorkspaceSize( | |||
| 47 | uint64_t *workspaceSize, | 47 | uint64_t *workspaceSize, |
| 48 | aclOpExecutor **executor) | 48 | aclOpExecutor **executor) |
| 49 | ``` | 49 | ``` |
| 50 | + | ||
| 50 | ```Cpp | 51 | ```Cpp |
| 51 | aclnnStatus aclnnFakeQuantPerTensorAffineCachemask( | 52 | aclnnStatus aclnnFakeQuantPerTensorAffineCachemask( |
| 52 | void *workspace, | 53 | void *workspace, |
| @@ -261,12 +262,13 @@ aclnnStatus aclnnFakeQuantPerTensorAffineCachemask( | |||
| 261 | </tr> | 262 | </tr> |
| 262 | </tbody> | 263 | </tbody> |
| 263 | </table> | 264 | </table> |
| 264 | - | 265 | + |
| 265 | - **返回值:** | 266 | - **返回值:** |
| 266 | 267 | ||
| 267 | aclnnStatus:返回状态码,具体参见[aclnn返回码](../../../docs/zh/context/aclnn返回码.md)。 | 268 | aclnnStatus:返回状态码,具体参见[aclnn返回码](../../../docs/zh/context/aclnn返回码.md)。 |
| 268 | 269 | ||
| 269 | ## 约束说明 | 270 | ## 约束说明 |
| 271 | + | ||
| 270 | - 确定性计算: | 272 | - 确定性计算: |
| 271 | - aclnnFakeQuantPerTensorAffineCachemask默认确定性实现。 | 273 | - aclnnFakeQuantPerTensorAffineCachemask默认确定性实现。 |
| 272 | 274 | ||
| @@ -430,4 +432,3 @@ int main() { | |||
| 430 | return 0; | 432 | return 0; |
| 431 | } | 433 | } |
| 432 | ``` | 434 | ``` |
| 433 | - | ||
| @@ -155,4 +155,4 @@ | |||
| 155 | | 调用方式 | 样例代码 | 说明 | | 155 | | 调用方式 | 样例代码 | 说明 | |
| 156 | | ---------------- | --------------------------- | --------------------------------------------------- | | 156 | | ---------------- | --------------------------- | --------------------------------------------------- | |
| 157 | | aclnn接口 | [test_aclnn_flat_quant](examples/test_aclnn_flat_quant.cpp) | 通过[aclnnFlatQuant](docs/aclnnFlatQuant.md)接口方式调用FlatQuant算子。 | | 157 | | aclnn接口 | [test_aclnn_flat_quant](examples/test_aclnn_flat_quant.cpp) | 通过[aclnnFlatQuant](docs/aclnnFlatQuant.md)接口方式调用FlatQuant算子。 | |
| 158 | -| 图模式 | - | 通过[算子IR](op_graph/flat_quant_proto.h)构图方式调用FlatQuant算子。 | | 158 | +| 图模式 | - | 通过[算子IR](op_graph/flat_quant_proto.h)构图方式调用FlatQuant算子。 | |
| @@ -271,8 +271,7 @@ aclnnStatus aclnnFlatQuant( | |||
| 271 | <td>out的数据类型为INT32时,x的shape尾轴不是out的shape尾轴大小的8倍,或者x与out的shape的非尾轴的大小不一致。</td> | 271 | <td>out的数据类型为INT32时,x的shape尾轴不是out的shape尾轴大小的8倍,或者x与out的shape的非尾轴的大小不一致。</td> |
| 272 | </tr> | 272 | </tr> |
| 273 | </tbody> | 273 | </tbody> |
| 274 | - | 274 | + </table> |
| 275 | -</table> | ||
| 276 | 275 | ||
| 277 | ## aclnnFlatQuant | 276 | ## aclnnFlatQuant |
| 278 | 277 | ||
| @@ -504,4 +503,4 @@ int main() | |||
| 504 | aclFinalize(); | 503 | aclFinalize(); |
| 505 | return 0; | 504 | return 0; |
| 506 | } | 505 | } |
| 507 | -``` | 506 | +``` |
| @@ -85,6 +85,7 @@ | |||
| 85 | </tbody></table> | 85 | </tbody></table> |
| 86 | 86 | ||
| 87 | - Kirin X90/Kirin 9030 处理器系列产品: 不支持BFLOAT16。 | 87 | - Kirin X90/Kirin 9030 处理器系列产品: 不支持BFLOAT16。 |
| 88 | + | ||
| 88 | ## 约束说明 | 89 | ## 约束说明 |
| 89 | 90 | ||
| 90 | - 输入`scale`与输入`offset`的数据类型一致。 | 91 | - 输入`scale`与输入`offset`的数据类型一致。 |
| @@ -97,4 +98,4 @@ | |||
| 97 | | 调用方式 | 样例代码 | 说明 | | 98 | | 调用方式 | 样例代码 | 说明 | |
| 98 | | ---------------- | --------------------------- | --------------------------------------------------- | | 99 | | ---------------- | --------------------------- | --------------------------------------------------- | |
| 99 | | aclnn接口 | [test_aclnn_group_quant](examples/test_aclnn_group_quant.cpp) | 通过[aclnnGroupQuant](docs/aclnnGroupQuant.md)接口方式调用GroupQuant算子。 | | 100 | | aclnn接口 | [test_aclnn_group_quant](examples/test_aclnn_group_quant.cpp) | 通过[aclnnGroupQuant](docs/aclnnGroupQuant.md)接口方式调用GroupQuant算子。 | |
| 100 | -| 图模式 | - | 通过[算子IR](op_graph/group_quant_proto.h)构图方式调用GroupQuant算子。 | | 101 | +| 图模式 | - | 通过[算子IR](op_graph/group_quant_proto.h)构图方式调用GroupQuant算子。 | |
| @@ -76,7 +76,7 @@ aclnnStatus aclnnGroupQuant( | |||
| 76 | <td>x(aclTensor*)</td> | 76 | <td>x(aclTensor*)</td> |
| 77 | <td>输入</td> | 77 | <td>输入</td> |
| 78 | <td>表示需要执行量化的输入,对应公式中的`x`。</td> | 78 | <td>表示需要执行量化的输入,对应公式中的`x`。</td> |
| 79 | - <td><ul><li>支持空Tensor。</li><li>如果`dstType`为3(INT32),shape的最后一维需要能被8整除。<li>如果`dstType`为29(INT4),shape的最后一维需要能被2整除。</li></ul></td> | 79 | + <td><ul><li>支持空Tensor。</li><li>如果`dstType`为3(INT32),shape的最后一维需要能被8整除。</li><li>如果`dstType`为29(INT4),shape的最后一维需要能被2整除。</li></ul></td> |
| 80 | <td>FLOAT32,FLOAT16,BFLOAT16</td> | 80 | <td>FLOAT32,FLOAT16,BFLOAT16</td> |
| 81 | <td>ND</td> | 81 | <td>ND</td> |
| 82 | <td>2</td> | 82 | <td>2</td> |
| @@ -29,7 +29,6 @@ | |||
| 29 | y = cast\_to\_[HiF8/FP8](input/scale) | 29 | y = cast\_to\_[HiF8/FP8](input/scale) |
| 30 | $$ | 30 | $$ |
| 31 | 31 | ||
| 32 | - | ||
| 33 | ## 参数说明 | 32 | ## 参数说明 |
| 34 | 33 | ||
| 35 | <table style="undefined;table-layout: fixed; width: 1401px"><colgroup> | 34 | <table style="undefined;table-layout: fixed; width: 1401px"><colgroup> |
| @@ -125,4 +124,4 @@ | |||
| 125 | | 调用方式 | 样例代码 | 说明 | | 124 | | 调用方式 | 样例代码 | 说明 | |
| 126 | | ---------------- | --------------------------- | --------------------------------------------------- | | 125 | | ---------------- | --------------------------- | --------------------------------------------------- | |
| 127 | | aclnn接口 | [test_aclnn_grouped_dynamic_block_quant](examples/arch35/test_aclnn_grouped_dynamic_block_quant.cpp) | 通过[aclnnGroupedDynamicBlockQuant](docs/aclnnGroupedDynamicBlockQuant.md)接口方式调用GroupedDynamicBlockQuant算子。 | | 126 | | aclnn接口 | [test_aclnn_grouped_dynamic_block_quant](examples/arch35/test_aclnn_grouped_dynamic_block_quant.cpp) | 通过[aclnnGroupedDynamicBlockQuant](docs/aclnnGroupedDynamicBlockQuant.md)接口方式调用GroupedDynamicBlockQuant算子。 | |
| 128 | -| 图模式 | - | 通过[算子IR](op_graph/grouped_dynamic_block_quant_proto.h)构图方式调用GroupedDynamicBlockQuant算子。 | | 127 | +| 图模式 | - | 通过[算子IR](op_graph/grouped_dynamic_block_quant_proto.h)构图方式调用GroupedDynamicBlockQuant算子。 | |
| @@ -28,7 +28,7 @@ | |||
| 28 | P_i = cast\_to\_dst\_type(V_i/mxscale, round\_mode), \space i\space from\space 1\space to\space blocksize \tag{3} | 28 | P_i = cast\_to\_dst\_type(V_i/mxscale, round\_mode), \space i\space from\space 1\space to\space blocksize \tag{3} |
| 29 | $$ | 29 | $$ |
| 30 | 30 | ||
| 31 | - 量化后的P<sub>i</sub>按对应的x<sub>i</sub>的位置组成输出y,mxscale_pre按对应的groupIndex分组,分组内第一个维度pad为偶数,组成输出mxscale。 | 31 | + 量化后的P<sub>i</sub>按对应的x<sub>i</sub>的位置组成输出y,mxscale_pre按对应的groupIndex分组,分组内第一个维度pad为偶数,组成输出mxscale。 |
| 32 | 32 | ||
| 33 | - emax: 对应数据类型的最大正则数的指数位。 | 33 | - emax: 对应数据类型的最大正则数的指数位。 |
| 34 | 34 | ||
| @@ -37,7 +37,6 @@ | |||
| 37 | | FLOAT8_E4M3FN | 8 | | 37 | | FLOAT8_E4M3FN | 8 | |
| 38 | | FLOAT8_E5M2 | 15 | | 38 | | FLOAT8_E5M2 | 15 | |
| 39 | 39 | ||
| 40 | - | ||
| 41 | ## 参数说明 | 40 | ## 参数说明 |
| 42 | 41 | ||
| 43 | <table style="undefined;table-layout: fixed; width: 980px"><colgroup> | 42 | <table style="undefined;table-layout: fixed; width: 980px"><colgroup> |
| @@ -110,8 +109,9 @@ | |||
| 110 | ## 约束说明 | 109 | ## 约束说明 |
| 111 | 110 | ||
| 112 | 无 | 111 | 无 |
| 112 | + | ||
| 113 | ## 调用说明 | 113 | ## 调用说明 |
| 114 | 114 | ||
| 115 | | 调用方式 | 调用样例 | 说明 | | 115 | | 调用方式 | 调用样例 | 说明 | |
| 116 | |--------------|------------------------------------------------------------------------|--------------------------------------------------------------| | 116 | |--------------|------------------------------------------------------------------------|--------------------------------------------------------------| |
| 117 | -| aclnn调用 | [test_aclnn_grouped_dynamic_mx_quant](./examples/test_aclnn_grouped_dynamic_mx_quant.cpp) | 通过[aclnnGroupedDynamicMxQuant](./docs/aclnnGroupedDynamicMxQuant.md)接口方式调用GroupedDynamicMxQuant算子。 | | 117 | +| aclnn调用 | [test_aclnn_grouped_dynamic_mx_quant](./examples/arch35/test_aclnn_grouped_dynamic_mx_quant.cpp) | 通过[aclnnGroupedDynamicMxQuant](./docs/aclnnGroupedDynamicMxQuant.md)接口方式调用GroupedDynamicMxQuant算子。 | |
| @@ -1,6 +1,6 @@ | |||
| 1 | # IFMR | 1 | # IFMR |
| 2 | 2 | ||
| 3 | -## 产品支持情况 | 3 | +## 产品支持情况 |
| 4 | 4 | ||
| 5 | | 产品 | 是否支持 | | 5 | | 产品 | 是否支持 | |
| 6 | | ---- | :----:| | 6 | | ---- | :----:| |
| @@ -127,4 +127,4 @@ | |||
| 127 | 127 | ||
| 128 | | 调用方式 | 样例代码 | 说明 | | 128 | | 调用方式 | 样例代码 | 说明 | |
| 129 | | ---------------- | --------------------------- | --------------------------------------------------- | | 129 | | ---------------- | --------------------------- | --------------------------------------------------- | |
| 130 | -| 图模式 | [test_geir_ifmr](./examples/test_geir_ifmr.cpp) | 通过[算子IR](./op_graph/ifmr_proto.h)构图方式调用IFMR算子。 | | 130 | +| 图模式 | [test_geir_ifmr](./examples/test_geir_ifmr.cpp) | 通过[算子IR](./op_graph/ifmr_proto.h)构图方式调用IFMR算子。 | |
| @@ -23,7 +23,6 @@ | |||
| 23 | out=round((x/scales)+zeroPoints) | 23 | out=round((x/scales)+zeroPoints) |
| 24 | $$ | 24 | $$ |
| 25 | 25 | ||
| 26 | - | ||
| 27 | ## 参数说明 | 26 | ## 参数说明 |
| 28 | 27 | ||
| 29 | <table style="undefined;table-layout: fixed; width: 820px"><colgroup> | 28 | <table style="undefined;table-layout: fixed; width: 820px"><colgroup> |
| @@ -72,9 +71,8 @@ | |||
| 72 | </tr> | 71 | </tr> |
| 73 | </tbody></table> | 72 | </tbody></table> |
| 74 | 73 | ||
| 75 | - | ||
| 76 | ## 调用说明 | 74 | ## 调用说明 |
| 77 | 75 | ||
| 78 | | 调用方式 | 样例代码 | 说明 | | 76 | | 调用方式 | 样例代码 | 说明 | |
| 79 | | ---------------- | --------------------------- | --------------------------------------------------- | | 77 | | ---------------- | --------------------------- | --------------------------------------------------- | |
| 80 | -| aclnn接口 | [test_aclnn_quantize.cpp](examples/test_aclnn_quantize.cpp) | 通过[aclnnQuantize.md](docs/aclnnQuantize.md)接口方式调用算子。 | | 78 | +| aclnn接口 | [test_aclnn_quantize.cpp](examples/test_aclnn_quantize.cpp) | 通过[aclnnQuantize.md](docs/aclnnQuantize.md)接口方式调用算子。 | |
| @@ -201,7 +201,8 @@ aclnnStatus aclnnQuantize( | |||
| 201 | <td>输入axis指定的轴超出输入x的维度数。</td> | 201 | <td>输入axis指定的轴超出输入x的维度数。</td> |
| 202 | </tr> | 202 | </tr> |
| 203 | <tr> | 203 | <tr> |
| 204 | - <td>输入dtype不在支持的范围之内。</tr> | 204 | + <td>输入dtype不在支持的范围之内。</td> |
| 205 | + </tr> | ||
| 205 | <tr> | 206 | <tr> |
| 206 | <td>输入scales和zeroPoints的size不相等。</td> | 207 | <td>输入scales和zeroPoints的size不相等。</td> |
| 207 | </tr> | 208 | </tr> |
| @@ -424,4 +425,4 @@ int main() | |||
| 424 | aclFinalize(); | 425 | aclFinalize(); |
| 425 | return 0; | 426 | return 0; |
| 426 | } | 427 | } |
| 427 | -``` | 428 | +``` |
| @@ -124,6 +124,7 @@ | |||
| 124 | <td>STRING</td> | 124 | <td>STRING</td> |
| 125 | <td>-</td> | 125 | <td>-</td> |
| 126 | </tr> | 126 | </tr> |
| 127 | + <tr> | ||
| 127 | <td>groupListType</td> | 128 | <td>groupListType</td> |
| 128 | <td>属性</td> | 129 | <td>属性</td> |
| 129 | <td><ul><li>用户必须传参。</li><li>0表示cumsum模式、1表示count模式。当前仅支持0 cumsum模式,1 count模式。</li></ul></td> | 130 | <td><ul><li>用户必须传参。</li><li>0表示cumsum模式、1表示count模式。当前仅支持0 cumsum模式,1 count模式。</li></ul></td> |
| @@ -137,7 +138,6 @@ | |||
| 137 | <td>INT64_T</td> | 138 | <td>INT64_T</td> |
| 138 | <td>-</td> | 139 | <td>-</td> |
| 139 | </tr> | 140 | </tr> |
| 140 | - <tr> | ||
| 141 | <tr> | 141 | <tr> |
| 142 | <td>yOut</td> | 142 | <td>yOut</td> |
| 143 | <td>输出</td> | 143 | <td>输出</td> |
| @@ -155,6 +155,7 @@ | |||
| 155 | </tbody></table> | 155 | </tbody></table> |
| 156 | 156 | ||
| 157 | - Kirin X90/Kirin 9030 处理器系列产品: `x`数据类型不支持BFLOAT16;`y`数据类型不支持INT4。 | 157 | - Kirin X90/Kirin 9030 处理器系列产品: `x`数据类型不支持BFLOAT16;`y`数据类型不支持INT4。 |
| 158 | + | ||
| 158 | ## 约束说明 | 159 | ## 约束说明 |
| 159 | 160 | ||
| 160 | 无 | 161 | 无 |
| @@ -152,7 +152,7 @@ aclnnStatus aclnnSwiGluQuantV2( | |||
| 152 | <td>groupIndexOptional(aclTensor*)</td> | 152 | <td>groupIndexOptional(aclTensor*)</td> |
| 153 | <td>输入</td> | 153 | <td>输入</td> |
| 154 | <td>MoE分组需要的group_index,公式中的group_index。</td> | 154 | <td>MoE分组需要的group_index,公式中的group_index。</td> |
| 155 | - <td>shape支持[G, ],group_index内元素要求为非递减,且最大值不得超过输入x的除最后一维之外的所有维度大小之积;G的值不得超过输入x的除最后一维之外的所有维度大小之积。</li></td> | 155 | + <td>shape支持[G, ],group_index内元素要求为非递减,且最大值不得超过输入x的除最后一维之外的所有维度大小之积;G的值不得超过输入x的除最后一维之外的所有维度大小之积。</td> |
| 156 | <td>INT32</td> | 156 | <td>INT32</td> |
| 157 | <td>ND</td> | 157 | <td>ND</td> |
| 158 | <td>-</td> | 158 | <td>-</td> |
| @@ -1,3 +1,3 @@ | |||
| 1 | # TransQuantParam | 1 | # TransQuantParam |
| 2 | 2 | ||
| 3 | -本目录仅包含TransQuantParam算子对应的aclnn接口;如您想要贡献该算子的AscendC实现,请参考[贡献流程](../../CONTRIBUTING.md)。 | 3 | +本目录仅包含TransQuantParam算子对应的aclnn接口;如您想要贡献该算子的AscendC实现,请参考[贡献流程](../../CONTRIBUTING.md)。 |
| @@ -82,7 +82,7 @@ aclnnStatus aclnnTransQuantParam( | |||
| 82 | <td>scaleArray(float*)</td> | 82 | <td>scaleArray(float*)</td> |
| 83 | <td>输入</td> | 83 | <td>输入</td> |
| 84 | <td>表示指向存储scale数据的内存,对应公式中的`scale`。</td> | 84 | <td>表示指向存储scale数据的内存,对应公式中的`scale`。</td> |
| 85 | - <td>需要保证scale数据中不存在NaN和inf。</li></ul></td> | 85 | + <td>需要保证scale数据中不存在NaN和inf。</td> |
| 86 | <td>-</td> | 86 | <td>-</td> |
| 87 | <td>-</td> | 87 | <td>-</td> |
| 88 | <td>-</td> | 88 | <td>-</td> |
| @@ -92,7 +92,7 @@ aclnnStatus aclnnTransQuantParam( | |||
| 92 | <td>scaleSize(uint64_t)</td> | 92 | <td>scaleSize(uint64_t)</td> |
| 93 | <td>输入</td> | 93 | <td>输入</td> |
| 94 | <td>表示scale数据的数量。</td> | 94 | <td>表示scale数据的数量。</td> |
| 95 | - <td>需要自行保证`scaleSize`与`scaleArray`包含的元素个数相同。</li></ul></td> | 95 | + <td>需要自行保证`scaleSize`与`scaleArray`包含的元素个数相同。</td> |
| 96 | <td>-</td> | 96 | <td>-</td> |
| 97 | <td>-</td> | 97 | <td>-</td> |
| 98 | <td>-</td> | 98 | <td>-</td> |
| @@ -102,7 +102,7 @@ aclnnStatus aclnnTransQuantParam( | |||
| 102 | <td>offsetArray(float*)</td> | 102 | <td>offsetArray(float*)</td> |
| 103 | <td>输入</td> | 103 | <td>输入</td> |
| 104 | <td>表示指向存储offset数据的内存,对应公式中的`offset`。</td> | 104 | <td>表示指向存储offset数据的内存,对应公式中的`offset`。</td> |
| 105 | - <td>需要保证offset数据中不存在NaN和inf。</li></ul></td> | 105 | + <td>需要保证offset数据中不存在NaN和inf。</td> |
| 106 | <td>-</td> | 106 | <td>-</td> |
| 107 | <td>-</td> | 107 | <td>-</td> |
| 108 | <td>-</td> | 108 | <td>-</td> |
| @@ -112,7 +112,7 @@ aclnnStatus aclnnTransQuantParam( | |||
| 112 | <td>offsetSize(uint64_t)</td> | 112 | <td>offsetSize(uint64_t)</td> |
| 113 | <td>输入</td> | 113 | <td>输入</td> |
| 114 | <td>表示offset数据的数量。</td> | 114 | <td>表示offset数据的数量。</td> |
| 115 | - <td>需要自行保证`offsetSize`与`offsetArray`包含的元素个数相同。</li></ul></td> | 115 | + <td>需要自行保证`offsetSize`与`offsetArray`包含的元素个数相同。</td> |
| 116 | <td>-</td> | 116 | <td>-</td> |
| 117 | <td>-</td> | 117 | <td>-</td> |
| 118 | <td>-</td> | 118 | <td>-</td> |
| @@ -122,7 +122,7 @@ aclnnStatus aclnnTransQuantParam( | |||
| 122 | <td>quantParam(uint64_t**)</td> | 122 | <td>quantParam(uint64_t**)</td> |
| 123 | <td>输出</td> | 123 | <td>输出</td> |
| 124 | <td>表示指向存储转换得到的quantParam数据的内存的地址,对应公式中的`out`。</td> | 124 | <td>表示指向存储转换得到的quantParam数据的内存的地址,对应公式中的`out`。</td> |
| 125 | - <td>-</li></ul></td> | 125 | + <td>-</td> |
| 126 | <td>-</td> | 126 | <td>-</td> |
| 127 | <td>-</td> | 127 | <td>-</td> |
| 128 | <td>-</td> | 128 | <td>-</td> |
| @@ -132,7 +132,7 @@ aclnnStatus aclnnTransQuantParam( | |||
| 132 | <td>quantParamSize(uint64_t*)</td> | 132 | <td>quantParamSize(uint64_t*)</td> |
| 133 | <td>输出</td> | 133 | <td>输出</td> |
| 134 | <td>表示存储quantParam数据的数量。</td> | 134 | <td>表示存储quantParam数据的数量。</td> |
| 135 | - <td>需要自行保证`quantParamSize`与`quantParam`包含的元素个数相同。</li></ul></td> | 135 | + <td>需要自行保证`quantParamSize`与`quantParam`包含的元素个数相同。</td> |
| 136 | <td>-</td> | 136 | <td>-</td> |
| 137 | <td>-</td> | 137 | <td>-</td> |
| 138 | <td>-</td> | 138 | <td>-</td> |
| @@ -48,7 +48,6 @@ | |||
| 48 | out = (out\ \&\ 0x4000FFFFFFFF)\ |\ ((offset\ \&\ 0x1FF)\ll37) | 48 | out = (out\ \&\ 0x4000FFFFFFFF)\ |\ ((offset\ \&\ 0x1FF)\ll37) |
| 49 | $$ | 49 | $$ |
| 50 | 50 | ||
| 51 | - | ||
| 52 | ## 参数说明 | 51 | ## 参数说明 |
| 53 | 52 | ||
| 54 | <table style="undefined;table-layout: fixed; width: 1005px"><colgroup> | 53 | <table style="undefined;table-layout: fixed; width: 1005px"><colgroup> |
| @@ -117,4 +116,4 @@ | |||
| 117 | | ---------------- | --------------------------- | --------------------------------------------------- | | 116 | | ---------------- | --------------------------- | --------------------------------------------------- | |
| 118 | | aclnn接口 | [test_aclnn_trans_quant_param_v2](examples/test_aclnn_trans_quant_param_v2.cpp) | 通过[aclnnTransQuantParamV2](docs/aclnnTransQuantParamV2.md)接口方式调用TransQuantParamV2算子。 | | 117 | | aclnn接口 | [test_aclnn_trans_quant_param_v2](examples/test_aclnn_trans_quant_param_v2.cpp) | 通过[aclnnTransQuantParamV2](docs/aclnnTransQuantParamV2.md)接口方式调用TransQuantParamV2算子。 | |
| 119 | | aclnn接口 | [test_aclnn_trans_quant_param_v3](examples/test_aclnn_trans_quant_param_v3.cpp) | 通过[aclnnTransQuantParamV3](docs/aclnnTransQuantParamV3.md)接口方式调用TransQuantParamV2算子。 | | 118 | | aclnn接口 | [test_aclnn_trans_quant_param_v3](examples/test_aclnn_trans_quant_param_v3.cpp) | 通过[aclnnTransQuantParamV3](docs/aclnnTransQuantParamV3.md)接口方式调用TransQuantParamV2算子。 | |
| 120 | -| 图模式 | - | 通过[算子IR](op_graph/trans_quant_param_v2_proto.h)构图方式调用TransQuantParamV2算子。 | | 119 | +| 图模式 | - | 通过[算子IR](op_graph/trans_quant_param_v2_proto.h)构图方式调用TransQuantParamV2算子。 | |
| @@ -65,7 +65,6 @@ aclnnStatus aclnnTransQuantParamV2( | |||
| 65 | 65 | ||
| 66 | - **参数说明:** | 66 | - **参数说明:** |
| 67 | 67 | ||
| 68 | - | ||
| 69 | <table style="undefined;table-layout: fixed; width: 1550px"><colgroup> | 68 | <table style="undefined;table-layout: fixed; width: 1550px"><colgroup> |
| 70 | <col style="width: 170px"> | 69 | <col style="width: 170px"> |
| 71 | <col style="width: 120px"> | 70 | <col style="width: 120px"> |
| @@ -140,7 +139,6 @@ aclnnStatus aclnnTransQuantParamV2( | |||
| 140 | </tr> | 139 | </tr> |
| 141 | </tbody> | 140 | </tbody> |
| 142 | </table> | 141 | </table> |
| 143 | - | ||
| 144 | 142 | ||
| 145 | - **返回值:** | 143 | - **返回值:** |
| 146 | 144 | ||
| @@ -73,7 +73,6 @@ aclnnStatus aclnnTransQuantParamV3( | |||
| 73 | 73 | ||
| 74 | - **参数说明:** | 74 | - **参数说明:** |
| 75 | 75 | ||
| 76 | - | ||
| 77 | <table style="undefined;table-layout: fixed; width: 1550px"><colgroup> | 76 | <table style="undefined;table-layout: fixed; width: 1550px"><colgroup> |
| 78 | <col style="width: 170px"> | 77 | <col style="width: 170px"> |
| 79 | <col style="width: 120px"> | 78 | <col style="width: 120px"> |
| @@ -192,7 +191,7 @@ aclnnStatus aclnnTransQuantParamV3( | |||
| 192 | <td>scale、offset的shape不在支持的范围内。</td> | 191 | <td>scale、offset的shape不在支持的范围内。</td> |
| 193 | </tr> | 192 | </tr> |
| 194 | <tr> | 193 | <tr> |
| 195 | - <td>roundMode的值不在支持的范围内。</tr> | 194 | + <td>roundMode的值不在支持的范围内。</td></tr> |
| 196 | </tbody></table> | 195 | </tbody></table> |
| 197 | 196 | ||
| 198 | ## aclnnTransQuantParamV3 | 197 | ## aclnnTransQuantParamV3 |