已合并
【feat】: 根据环境变量去掉div高精度 #2120
WangYanMale创建于 14 天前
【feat】: 根据环境变量去掉div高精度 #2120
已合并
共 6 个文件变更+147-5
| @@ -712,8 +712,7 @@ bool ShouldSkipGraph(optimize::GraphPropertiesCache &cache, const AscGraph &asc_ | |||
| 712 | 712 | ||
| 713 | Status IsAllNodesInBlacklist(const AscGraph &asc_graph, bool &result) { | 713 | Status IsAllNodesInBlacklist(const AscGraph &asc_graph, bool &result) { |
| 714 | const auto &blacklist2 = PreProcessConfig::Instance().GetImprovePrecisionBlacklist(); | 714 | const auto &blacklist2 = PreProcessConfig::Instance().GetImprovePrecisionBlacklist(); |
| 715 | - constexpr char kAllNodesType[] = "all"; | 715 | + const bool has_all = (blacklist2.find(PreProcessConfig::kAllNodeType) != blacklist2.end()); |
| 716 | - const bool has_all = (blacklist2.find(kAllNodesType) != blacklist2.end()); | ||
| 717 | result = true; | 716 | result = true; |
| 718 | for (const auto &node : AscGraphUtils::GetComputeGraph(asc_graph)->GetAllNodes()) { | 717 | for (const auto &node : AscGraphUtils::GetComputeGraph(asc_graph)->GetAllNodes()) { |
| 719 | if (node->GetType() == af::ascir_op::Output::Type || node->GetType() == af::ascir_op::Data::Type) { | 718 | if (node->GetType() == af::ascir_op::Output::Type || node->GetType() == af::ascir_op::Data::Type) { |
| @@ -26,10 +26,19 @@ class PreProcessConfig { | |||
| 26 | return config; | 26 | return config; |
| 27 | } | 27 | } |
| 28 | 28 | ||
| 29 | + // 黑名单通配值,表示所有算子类型均命中 | ||
| 30 | + static constexpr char kAllNodeType[] = "all"; | ||
| 31 | + | ||
| 29 | const std::unordered_set<std::string> &GetImprovePrecisionBlacklist() const { | 32 | const std::unordered_set<std::string> &GetImprovePrecisionBlacklist() const { |
| 30 | return blacklist_; | 33 | return blacklist_; |
| 31 | } | 34 | } |
| 32 | 35 | ||
| 36 | + // 通用精度黑名单查询:配置 all 或命中指定算子类型时返回 true。 | ||
| 37 | + // 供 pre-process 精度提升 pass 与 codegen 高精度模式(如 Div)等场景统一使用。 | ||
| 38 | + bool IsInImprovePrecisionBlacklist(const std::string &op_type) const { | ||
| 39 | + return blacklist_.find(kAllNodeType) != blacklist_.end() || blacklist_.find(op_type) != blacklist_.end(); | ||
| 40 | + } | ||
| 41 | + | ||
| 33 | void Reset() { | 42 | void Reset() { |
| 34 | blacklist_.clear(); | 43 | blacklist_.clear(); |
| 35 | ParseBlacklist(); | 44 | ParseBlacklist(); |
| @@ -17,6 +17,7 @@ | |||
| 17 | 17 | ||
| 18 | 18 | ||
| 19 | 19 | ||
| 20 | + | ||
| 20 | 21 | ||
| 21 | using namespace std; | 22 | using namespace std; |
| 22 | using namespace ascir; | 23 | using namespace ascir; |
| @@ -147,6 +148,134 @@ TEST(CodegenKernel, MicroDivApiCall_Load_Div_Float_Store) { | |||
| 147 | EXPECT_EQ(result, std::string{"AscendC::MicroAPI::Div<float, &high_precision_div_mode>(vreg_1, vreg_0, p_reg);\n"}); | 148 | EXPECT_EQ(result, std::string{"AscendC::MicroAPI::Div<float, &high_precision_div_mode>(vreg_1, vreg_0, p_reg);\n"}); |
| 148 | } | 149 | } |
| 149 | 150 | ||
| 151 | +TEST(CodegenKernel, MicroDivApiCall_Load_Div_Float_EnhancePrecisionBlacklist) { | ||
| 152 | + setenv("AUTOFUSE_FLAGS", "--autofuse_enhance_precision_blacklist=Div", 1); | ||
| 153 | + af::pre_process::PreProcessConfig::Instance().Reset(); | ||
| 154 | + | ||
| 155 | + af::AscGraph graph("test_div_graph"); | ||
| 156 | + | ||
| 157 | + af::Expression Two = af::Symbol(2); | ||
| 158 | + af::Expression Three = af::Symbol(3); | ||
| 159 | + af::Expression Four = af::Symbol(4); | ||
| 160 | + | ||
| 161 | + auto s0 = af::Symbol(16); | ||
| 162 | + auto s1 = af::Symbol(8); | ||
| 163 | + auto s2 = af::Symbol(4); | ||
| 164 | + auto s3 = af::Symbol(2); | ||
| 165 | + auto z0 = graph.CreateAxis("z0", s0); | ||
| 166 | + auto z1 = graph.CreateAxis("z1", s1); | ||
| 167 | + auto z2 = graph.CreateAxis("z2", s2); | ||
| 168 | + auto z3 = graph.CreateAxis("z3", s3); | ||
| 169 | + | ||
| 170 | + af::ascir_op::Data x_op("x", graph); | ||
| 171 | + af::ascir_op::Load load_op("load"); | ||
| 172 | + af::ascir_op::Div div_op("div"); | ||
| 173 | + af::ascir_op::Store store_op("store"); | ||
| 174 | + | ||
| 175 | + graph.AddNode(load_op); | ||
| 176 | + graph.AddNode(div_op); | ||
| 177 | + graph.AddNode(store_op); | ||
| 178 | + | ||
| 179 | + load_op.x = x_op.y; | ||
| 180 | + load_op.attr.sched.axis = {z0.id, z1.id, z2.id, z3.id}; | ||
| 181 | + *load_op.y.axis = {z0.id, z1.id, z2.id, z3.id}; | ||
| 182 | + *load_op.y.repeats = {s0, s1, s2, s3}; | ||
| 183 | + *load_op.y.strides = {s1 * s2 * s3 * Four, s2 * s3 * Three, s3 * Two, One}; | ||
| 184 | + | ||
| 185 | + div_op.x1 = load_op.y; | ||
| 186 | + div_op.attr.sched.axis = {z0.id, z1.id, z2.id, z3.id}; | ||
| 187 | + *div_op.y.axis = {z0.id, z1.id, z2.id, z3.id}; | ||
| 188 | + *div_op.y.repeats = {s0, s1, s2, s3}; | ||
| 189 | + *div_op.y.strides = {s1 * s2 * s3 * Four, s2 * s3 * Three, s3 * Two, One}; | ||
| 190 | + | ||
| 191 | + store_op.x = div_op.y; | ||
| 192 | + store_op.ir_attr.SetOffset(af::Symbol(0)); | ||
| 193 | + *store_op.y.axis = {z0.id, z1.id, z2.id, z3.id}; | ||
| 194 | + *store_op.y.repeats = {s0, s1, s2, s3}; | ||
| 195 | + *store_op.y.strides = {s1 * s2 * s3 * Four, s2 * s3 * Three, s3 * Two, One}; | ||
| 196 | + | ||
| 197 | + auto load = graph.FindNode("load"); | ||
| 198 | + load->attr.api.compute_type = af::ComputeType::kComputeLoad; | ||
| 199 | + load->attr.api.type = af::ApiType::kAPITypeCompute; | ||
| 200 | + load->attr.api.unit = af::ComputeUnit::kUnitMTE2; | ||
| 201 | + load->attr.sched.loop_axis = z0.id; | ||
| 202 | + load->outputs[0].attr.vectorized_axis = {z1.id, z2.id, z3.id}; | ||
| 203 | + load->outputs[0].attr.vectorized_strides = {af::Symbol(8), af::Symbol(2), One}; | ||
| 204 | + load->outputs[0].attr.dtype = af::DT_FLOAT; | ||
| 205 | + load->outputs[0].attr.mem.position = af::Position::kPositionVecIn; | ||
| 206 | + load->outputs[0].attr.mem.tensor_id = 0; | ||
| 207 | + load->outputs[0].attr.mem.position = af::Position::kPositionVecIn; | ||
| 208 | + load->outputs[0].attr.mem.alloc_type = af::AllocType::kAllocTypeQueue; | ||
| 209 | + load->outputs[0].attr.que.id = 1; | ||
| 210 | + load->outputs[0].attr.opt.merge_scope = af::kIdNone; | ||
| 211 | + | ||
| 212 | + auto div_node = graph.FindNode("div"); | ||
| 213 | + div_node->attr.api.compute_type = af::ComputeType::kComputeLoad; | ||
| 214 | + div_node->attr.api.type = af::ApiType::kAPITypeCompute; | ||
| 215 | + div_node->attr.api.unit = af::ComputeUnit::kUnitMTE2; | ||
| 216 | + div_node->attr.sched.loop_axis = z0.id; | ||
| 217 | + div_node->outputs[0].attr.vectorized_axis = {z1.id, z2.id, z3.id}; | ||
| 218 | + div_node->outputs[0].attr.vectorized_strides = {af::Symbol(8), af::Symbol(2), One}; | ||
| 219 | + div_node->outputs[0].attr.dtype = af::DT_FLOAT; | ||
| 220 | + div_node->outputs[0].attr.mem.position = af::Position::kPositionVecIn; | ||
| 221 | + div_node->outputs[0].attr.mem.tensor_id = 1; | ||
| 222 | + div_node->outputs[0].attr.mem.position = af::Position::kPositionVecIn; | ||
| 223 | + div_node->outputs[0].attr.mem.alloc_type = af::AllocType::kAllocTypeQueue; | ||
| 224 | + div_node->outputs[0].attr.que.id = 2; | ||
| 225 | + div_node->outputs[0].attr.opt.merge_scope = af::kIdNone; | ||
| 226 | + | ||
| 227 | + auto store = graph.FindNode("store"); | ||
| 228 | + store->attr.api.compute_type = af::ComputeType::kComputeElewise; | ||
| 229 | + store->attr.api.type = af::ApiType::kAPITypeCompute; | ||
| 230 | + store->attr.api.unit = af::ComputeUnit::kUnitVector; | ||
| 231 | + store->attr.sched.loop_axis = z0.id; | ||
| 232 | + store->outputs[0].attr.vectorized_axis = {z1.id, z2.id, z3.id}; | ||
| 233 | + store->outputs[0].attr.vectorized_strides = {af::Symbol(8), af::Symbol(2), One}; | ||
| 234 | + store->outputs[0].attr.dtype = af::DT_FLOAT; | ||
| 235 | + store->outputs[0].attr.mem.position = af::Position::kPositionVecOut; | ||
| 236 | + store->outputs[0].attr.mem.tensor_id = 1; | ||
| 237 | + store->outputs[0].attr.mem.alloc_type = af::AllocType::kAllocTypeQueue; | ||
| 238 | + store->outputs[0].attr.que.id = 2; | ||
| 239 | + store->outputs[0].attr.opt.merge_scope = af::kIdNone; | ||
| 240 | + | ||
| 241 | + codegen::Tiler tiler; | ||
| 242 | + codegen::TPipe tpipe("tpipe", tiler); | ||
| 243 | + tpipe.AddTensor(load->outputs[0]); | ||
| 244 | + | ||
| 245 | + tiler.AddAxis(z0); | ||
| 246 | + tiler.AddAxis(z1); | ||
| 247 | + tiler.AddAxis(z2); | ||
| 248 | + tiler.AddAxis(z3); | ||
| 249 | + tiler.AddSizeVar(af::SizeVar(s0)); | ||
| 250 | + tiler.AddSizeVar(af::SizeVar(s1)); | ||
| 251 | + tiler.AddSizeVar(af::SizeVar(s2)); | ||
| 252 | + tiler.AddSizeVar(af::SizeVar(s3)); | ||
| 253 | + | ||
| 254 | + codegen::ApiTensor x1; | ||
| 255 | + x1.id = load->outputs[0].attr.mem.tensor_id; | ||
| 256 | + codegen::ApiTensor y1; | ||
| 257 | + y1.id = store->outputs[0].attr.mem.tensor_id; | ||
| 258 | + codegen::CallParam cp = {"p_reg", ""}; | ||
| 259 | + auto tensor_load = load->GetName() + "_" + load->GetOpDesc()->GetOutputNameByIndex(0); | ||
| 260 | + MicroApiTensor tensor1(load->outputs[0], tensor_load); | ||
| 261 | + auto tensor_store = store->GetName() + "_" + store->GetOpDesc()->GetOutputNameByIndex(0); | ||
| 262 | + MicroApiTensor tensor2(store->outputs[0], tensor_store); | ||
| 263 | + TensorManager tensor_mng; | ||
| 264 | + tensor_mng.AddTensor(tensor1); | ||
| 265 | + tensor_mng.AddTensor(tensor2); | ||
| 266 | + codegen::MicroDivApiCall call("Div"); | ||
| 267 | + EXPECT_EQ(call.Init(div_node), 0); | ||
| 268 | + call.AddInput(x1.id); | ||
| 269 | + call.AddOutput(y1.id); | ||
| 270 | + | ||
| 271 | + std::string result; | ||
| 272 | + call.Generate(tensor_mng, tpipe, cp, result); | ||
| 273 | + EXPECT_EQ(result, std::string{"AscendC::MicroAPI::Div<float>(vreg_1, vreg_0, p_reg);\n"}); | ||
| 274 | + | ||
| 275 | + unsetenv("AUTOFUSE_FLAGS"); | ||
| 276 | + af::pre_process::PreProcessConfig::Instance().Reset(); | ||
| 277 | +} | ||
| 278 | + | ||
| 150 | TEST(CodegenKernel, MicroDivApiCall_Load_Div_Half_Store) { | 279 | TEST(CodegenKernel, MicroDivApiCall_Load_Div_Half_Store) { |
| 151 | af::AscGraph graph("test_div_graph"); | 280 | af::AscGraph graph("test_div_graph"); |
| 152 | 281 | ||
| @@ -9,6 +9,7 @@ | |||
| 9 | 9 | ||
| 10 | 10 | ||
| 11 | 11 | ||
| 12 | + | ||
| 12 | 13 | ||
| 13 | namespace codegen { | 14 | namespace codegen { |
| 14 | Status MicroDivApiCall::Generate(const codegen::TensorManager &tensor_mng, [[maybe_unused]] const TPipe &tpipe, | 15 | Status MicroDivApiCall::Generate(const codegen::TensorManager &tensor_mng, [[maybe_unused]] const TPipe &tpipe, |
| @@ -26,7 +27,11 @@ Status MicroDivApiCall::Generate(const codegen::TensorManager &tensor_mng, [[may | |||
| 26 | static_cast<int32_t>(input_dtype)); | 27 | static_cast<int32_t>(input_dtype)); |
| 27 | ss << "AscendC::MicroAPI::" << this->api_name_; | 28 | ss << "AscendC::MicroAPI::" << this->api_name_; |
| 28 | if (input_dtype == ge::DT_FLOAT) { | 29 | if (input_dtype == ge::DT_FLOAT) { |
| 29 | - ss << "<" << input_dtype_name << ", &high_precision_div_mode" << ">"; | 30 | + ss << "<" << input_dtype_name; |
| 31 | + if (!af::pre_process::PreProcessConfig::Instance().IsInImprovePrecisionBlacklist(af::ascir_op::Div::Type)) { | ||
| 32 | + ss << ", &high_precision_div_mode"; | ||
| 33 | + } | ||
| 34 | + ss << ">"; | ||
| 30 | } | 35 | } |
| 31 | ss << "("; | 36 | ss << "("; |
| 32 | for (const auto &out_arg : this->outputs_) { | 37 | for (const auto &out_arg : this->outputs_) { |
| @@ -40,7 +40,7 @@ For PyTorch, AutoFuse is enabled by configuring the `ascendc` backend through `t | |||
| 40 | | `--enable_autofuse` | TensorFlow | Controls whether automatic fusion is enabled globally. Accepts `true` or `false`; `false` is the default. Other AutoFuse control options have no effect when it is disabled. | | 40 | | `--enable_autofuse` | TensorFlow | Controls whether automatic fusion is enabled globally. Accepts `true` or `false`; `false` is the default. Other AutoFuse control options have no effect when it is disabled. | |
| 41 | | `--autofuse_enable_pass` | TensorFlow | Enables specified extended fusion capabilities. Currently, `reduce` and `concat` are supported. Multiple values are separated by commas; the default is empty and extended fusion is disabled. The same value must not be configured with `--autofuse_disable_pass`. | | 41 | | `--autofuse_enable_pass` | TensorFlow | Enables specified extended fusion capabilities. Currently, `reduce` and `concat` are supported. Multiple values are separated by commas; the default is empty and extended fusion is disabled. The same value must not be configured with `--autofuse_disable_pass`. | |
| 42 | | `--autofuse_disable_pass` | TensorFlow | Disables specified extended fusion capabilities. Supports `reduce` and `concat`; multiple capabilities can be disabled by separating their values with commas. The default is empty. The same value must not be configured with `--autofuse_enable_pass`. | | 42 | | `--autofuse_disable_pass` | TensorFlow | Disables specified extended fusion capabilities. Supports `reduce` and `concat`; multiple capabilities can be disabled by separating their values with commas. The default is empty. The same value must not be configured with `--autofuse_enable_pass`. | |
| 43 | -| `--autofuse_enhance_precision_blacklist` | TensorFlow | Controls whether specified AscIR operator types skip precision enhancement. Accepts comma-separated AscIR operator type strings or `all`; default: empty. `Sum`, `Mean`, and `Prod` still require precision enhancement. | | 43 | +| `--autofuse_enhance_precision_blacklist` | TensorFlow, PyTorch | Controls whether specified AscIR operator types skip precision enhancement. Accepts comma-separated AscIR operator type strings or `all`; default: empty. `Sum`, `Mean`, and `Prod` still require precision enhancement. When `Div` or `all` is configured, `Div` operators are generated with the default precision mode instead of the high-precision mode, which may affect accuracy. | |
| 44 | | `--recomputation_threshold` | TensorFlow | Sets the automatic-fusion recomputation threshold. Accepts an integer from `0` to `255`; default: `1`. none. | | 44 | | `--recomputation_threshold` | TensorFlow | Sets the automatic-fusion recomputation threshold. Accepts an integer from `0` to `255`; default: `1`. none. | |
| 45 | | `--max_fusion_size` | TensorFlow | Sets the maximum number of nodes in a fused operator. Accepts `0` to the maximum `uint64_t` value; `0` disables fusion; the default is implementation-defined. none. | | 45 | | `--max_fusion_size` | TensorFlow | Sets the maximum number of nodes in a fused operator. Accepts `0` to the maximum `uint64_t` value; `0` disables fusion; the default is implementation-defined. none. | |
| 46 | | `--autofuse_enable_pgo` | TensorFlow, PyTorch | Enables PGO tuning by selecting better-performing Tiling through pre-run sampling. Accepts `true` or `false`; default: `false`. static graphs only, `mspti` is required, and the first configuration cannot be used with other Profiling features. | | 46 | | `--autofuse_enable_pgo` | TensorFlow, PyTorch | Enables PGO tuning by selecting better-performing Tiling through pre-run sampling. Accepts `true` or `false`; default: `false`. static graphs only, `mspti` is required, and the first configuration cannot be used with other Profiling features. | |
| @@ -44,7 +44,7 @@ export AUTOFUSE_FLAGS="--enable_autofuse=true" | |||
| 44 | | `--enable_autofuse` | TensorFlow | 控制整体自动融合功能是否开启。取值为 `true` 或 `false`,`false` 为默认值;未开启时,其他 AutoFuse 控制项均不生效。 | | 44 | | `--enable_autofuse` | TensorFlow | 控制整体自动融合功能是否开启。取值为 `true` 或 `false`,`false` 为默认值;未开启时,其他 AutoFuse 控制项均不生效。 | |
| 45 | | `--autofuse_enable_pass` | TensorFlow | 控制指定的扩展融合能力是否开启。目前支持 `reduce` 和 `concat`;多个取值使用英文逗号分隔,默认不配置,扩展融合默认关闭。不能与 `--autofuse_disable_pass` 配置相同取值。 | | 45 | | `--autofuse_enable_pass` | TensorFlow | 控制指定的扩展融合能力是否开启。目前支持 `reduce` 和 `concat`;多个取值使用英文逗号分隔,默认不配置,扩展融合默认关闭。不能与 `--autofuse_disable_pass` 配置相同取值。 | |
| 46 | | `--autofuse_disable_pass` | TensorFlow | 控制指定的扩展融合能力是否关闭。支持配置 `reduce`、`concat`,也可以使用英文逗号分隔,同时关闭多个扩展融合能力。默认不配置;不能与 `--autofuse_enable_pass` 配置相同取值。 | | 46 | | `--autofuse_disable_pass` | TensorFlow | 控制指定的扩展融合能力是否关闭。支持配置 `reduce`、`concat`,也可以使用英文逗号分隔,同时关闭多个扩展融合能力。默认不配置;不能与 `--autofuse_enable_pass` 配置相同取值。 | |
| 47 | -| `--autofuse_enhance_precision_blacklist` | TensorFlow | 控制指定 AscIR 算子类型是否跳过精度提升。取值为 AscIR 算子类型字符串,多个类型使用英文逗号分隔,也可配置为 `all`;默认值为空。`Sum`、`Mean`、`Prod` 不支持低精度类型,即使加入黑名单也仍会提升精度。 | | 47 | +| `--autofuse_enhance_precision_blacklist` | TensorFlow、PyTorch | 控制指定 AscIR 算子类型是否跳过精度提升。取值为 AscIR 算子类型字符串,多个类型使用英文逗号分隔,也可配置为 `all`;默认值为空。`Sum`、`Mean`、`Prod` 不支持低精度类型,即使加入黑名单也仍会提升精度。配置 `Div` 或 `all` 后,`Div` 算子在代码生成时关闭高精度模式,按默认精度生成,可能影响计算精度。 | |
| 48 | | `--recomputation_threshold` | TensorFlow | 设置自动融合重计算阈值。取值为 `0`~`255` 的整数,默认值为 `1`。 | | 48 | | `--recomputation_threshold` | TensorFlow | 设置自动融合重计算阈值。取值为 `0`~`255` 的整数,默认值为 `1`。 | |
| 49 | | `--max_fusion_size` | TensorFlow | 设置单个融合算子最多包含的节点数量。取值为 `0`~`uint64_t` 最大值,配置为 `0` 表示不融合,默认值由实现决定。 | | 49 | | `--max_fusion_size` | TensorFlow | 设置单个融合算子最多包含的节点数量。取值为 `0`~`uint64_t` 最大值,配置为 `0` 表示不融合,默认值由实现决定。 | |
| 50 | | `--autofuse_enable_pgo` | TensorFlow、PyTorch | 开启 PGO 调优,通过预先上板采样选择性能更优的 Tiling。取值为 `true` 或 `false`,默认值为 `false`。仅支持静态图调优,需要准备对应版本的 `mspti`;首次配置时不能与其他 Profiling 功能同时开启。 | | 50 | | `--autofuse_enable_pgo` | TensorFlow、PyTorch | 开启 PGO 调优,通过预先上板采样选择性能更优的 Tiling。取值为 `true` 或 `false`,默认值为 `false`。仅支持静态图调优,需要准备对应版本的 `mspti`;首次配置时不能与其他 Profiling 功能同时开启。 | |