已合并
【feat】: 根据环境变量去掉div高精度 #2120
【feat】: 根据环境变量去掉div高精度 #2120
已合并
WangYanMale创建于 14 天前
共 6 个文件变更+147-5
@@ -712,8 +712,7 @@ bool ShouldSkipGraph(optimize::GraphPropertiesCache &cache, const AscGraph &asc_
712 712 
713Status IsAllNodesInBlacklist(const AscGraph &asc_graph, bool &result) {713Status IsAllNodesInBlacklist(const AscGraph &asc_graph, bool &result) {
714 const auto &blacklist2 = PreProcessConfig::Instance().GetImprovePrecisionBlacklist();714 const auto &blacklist2 = PreProcessConfig::Instance().GetImprovePrecisionBlacklist();
715- constexpr char kAllNodesType[] = "all";715+ const bool has_all = (blacklist2.find(PreProcessConfig::kAllNodeType) != blacklist2.end());
716- const bool has_all = (blacklist2.find(kAllNodesType) != blacklist2.end());
717 result = true;716 result = true;
718 for (const auto &node : AscGraphUtils::GetComputeGraph(asc_graph)->GetAllNodes()) {717 for (const auto &node : AscGraphUtils::GetComputeGraph(asc_graph)->GetAllNodes()) {
719 if (node->GetType() == af::ascir_op::Output::Type || node->GetType() == af::ascir_op::Data::Type) {718 if (node->GetType() == af::ascir_op::Output::Type || node->GetType() == af::ascir_op::Data::Type) {
@@ -26,10 +26,19 @@ class PreProcessConfig {
26 return config;26 return config;
27 }27 }
28 28 
29+ // 黑名单通配值,表示所有算子类型均命中
30+ static constexpr char kAllNodeType[] = "all";
31+ 
29 const std::unordered_set<std::string> &GetImprovePrecisionBlacklist() const {32 const std::unordered_set<std::string> &GetImprovePrecisionBlacklist() const {
30 return blacklist_;33 return blacklist_;
31 }34 }
32 35 
36+ // 通用精度黑名单查询:配置 all 或命中指定算子类型时返回 true。
37+ // 供 pre-process 精度提升 pass 与 codegen 高精度模式(如 Div)等场景统一使用。
38+ bool IsInImprovePrecisionBlacklist(const std::string &op_type) const {
39+ return blacklist_.find(kAllNodeType) != blacklist_.end() || blacklist_.find(op_type) != blacklist_.end();
40+ }
41+ 
33 void Reset() {42 void Reset() {
34 blacklist_.clear();43 blacklist_.clear();
35 ParseBlacklist();44 ParseBlacklist();
@@ -17,6 +17,7 @@
17#include "codegen_kernel.h"17#include "codegen_kernel.h"
18#include "micro_api_call_factory.h"18#include "micro_api_call_factory.h"
19#include "micro_div_api_call.h"19#include "micro_div_api_call.h"
20+#include "optimize/pre_process/pre_process_config.h"
20 21 
21using namespace std;22using namespace std;
22using namespace ascir;23using namespace ascir;
@@ -147,6 +148,134 @@ TEST(CodegenKernel, MicroDivApiCall_Load_Div_Float_Store) {
147 EXPECT_EQ(result, std::string{"AscendC::MicroAPI::Div<float, &high_precision_div_mode>(vreg_1, vreg_0, p_reg);\n"});148 EXPECT_EQ(result, std::string{"AscendC::MicroAPI::Div<float, &high_precision_div_mode>(vreg_1, vreg_0, p_reg);\n"});
148}149}
149 150 
151+TEST(CodegenKernel, MicroDivApiCall_Load_Div_Float_EnhancePrecisionBlacklist) {
152+ setenv("AUTOFUSE_FLAGS", "--autofuse_enhance_precision_blacklist=Div", 1);
153+ af::pre_process::PreProcessConfig::Instance().Reset();
154+ 
155+ af::AscGraph graph("test_div_graph");
156+ 
157+ af::Expression Two = af::Symbol(2);
158+ af::Expression Three = af::Symbol(3);
159+ af::Expression Four = af::Symbol(4);
160+ 
161+ auto s0 = af::Symbol(16);
162+ auto s1 = af::Symbol(8);
163+ auto s2 = af::Symbol(4);
164+ auto s3 = af::Symbol(2);
165+ auto z0 = graph.CreateAxis("z0", s0);
166+ auto z1 = graph.CreateAxis("z1", s1);
167+ auto z2 = graph.CreateAxis("z2", s2);
168+ auto z3 = graph.CreateAxis("z3", s3);
169+ 
170+ af::ascir_op::Data x_op("x", graph);
171+ af::ascir_op::Load load_op("load");
172+ af::ascir_op::Div div_op("div");
173+ af::ascir_op::Store store_op("store");
174+ 
175+ graph.AddNode(load_op);
176+ graph.AddNode(div_op);
177+ graph.AddNode(store_op);
178+ 
179+ load_op.x = x_op.y;
180+ load_op.attr.sched.axis = {z0.id, z1.id, z2.id, z3.id};
181+ *load_op.y.axis = {z0.id, z1.id, z2.id, z3.id};
182+ *load_op.y.repeats = {s0, s1, s2, s3};
183+ *load_op.y.strides = {s1 * s2 * s3 * Four, s2 * s3 * Three, s3 * Two, One};
184+ 
185+ div_op.x1 = load_op.y;
186+ div_op.attr.sched.axis = {z0.id, z1.id, z2.id, z3.id};
187+ *div_op.y.axis = {z0.id, z1.id, z2.id, z3.id};
188+ *div_op.y.repeats = {s0, s1, s2, s3};
189+ *div_op.y.strides = {s1 * s2 * s3 * Four, s2 * s3 * Three, s3 * Two, One};
190+ 
191+ store_op.x = div_op.y;
192+ store_op.ir_attr.SetOffset(af::Symbol(0));
193+ *store_op.y.axis = {z0.id, z1.id, z2.id, z3.id};
194+ *store_op.y.repeats = {s0, s1, s2, s3};
195+ *store_op.y.strides = {s1 * s2 * s3 * Four, s2 * s3 * Three, s3 * Two, One};
196+ 
197+ auto load = graph.FindNode("load");
198+ load->attr.api.compute_type = af::ComputeType::kComputeLoad;
199+ load->attr.api.type = af::ApiType::kAPITypeCompute;
200+ load->attr.api.unit = af::ComputeUnit::kUnitMTE2;
201+ load->attr.sched.loop_axis = z0.id;
202+ load->outputs[0].attr.vectorized_axis = {z1.id, z2.id, z3.id};
203+ load->outputs[0].attr.vectorized_strides = {af::Symbol(8), af::Symbol(2), One};
204+ load->outputs[0].attr.dtype = af::DT_FLOAT;
205+ load->outputs[0].attr.mem.position = af::Position::kPositionVecIn;
206+ load->outputs[0].attr.mem.tensor_id = 0;
207+ load->outputs[0].attr.mem.position = af::Position::kPositionVecIn;
208+ load->outputs[0].attr.mem.alloc_type = af::AllocType::kAllocTypeQueue;
209+ load->outputs[0].attr.que.id = 1;
210+ load->outputs[0].attr.opt.merge_scope = af::kIdNone;
211+ 
212+ auto div_node = graph.FindNode("div");
213+ div_node->attr.api.compute_type = af::ComputeType::kComputeLoad;
214+ div_node->attr.api.type = af::ApiType::kAPITypeCompute;
215+ div_node->attr.api.unit = af::ComputeUnit::kUnitMTE2;
216+ div_node->attr.sched.loop_axis = z0.id;
217+ div_node->outputs[0].attr.vectorized_axis = {z1.id, z2.id, z3.id};
218+ div_node->outputs[0].attr.vectorized_strides = {af::Symbol(8), af::Symbol(2), One};
219+ div_node->outputs[0].attr.dtype = af::DT_FLOAT;
220+ div_node->outputs[0].attr.mem.position = af::Position::kPositionVecIn;
221+ div_node->outputs[0].attr.mem.tensor_id = 1;
222+ div_node->outputs[0].attr.mem.position = af::Position::kPositionVecIn;
223+ div_node->outputs[0].attr.mem.alloc_type = af::AllocType::kAllocTypeQueue;
224+ div_node->outputs[0].attr.que.id = 2;
225+ div_node->outputs[0].attr.opt.merge_scope = af::kIdNone;
226+ 
227+ auto store = graph.FindNode("store");
228+ store->attr.api.compute_type = af::ComputeType::kComputeElewise;
229+ store->attr.api.type = af::ApiType::kAPITypeCompute;
230+ store->attr.api.unit = af::ComputeUnit::kUnitVector;
231+ store->attr.sched.loop_axis = z0.id;
232+ store->outputs[0].attr.vectorized_axis = {z1.id, z2.id, z3.id};
233+ store->outputs[0].attr.vectorized_strides = {af::Symbol(8), af::Symbol(2), One};
234+ store->outputs[0].attr.dtype = af::DT_FLOAT;
235+ store->outputs[0].attr.mem.position = af::Position::kPositionVecOut;
236+ store->outputs[0].attr.mem.tensor_id = 1;
237+ store->outputs[0].attr.mem.alloc_type = af::AllocType::kAllocTypeQueue;
238+ store->outputs[0].attr.que.id = 2;
239+ store->outputs[0].attr.opt.merge_scope = af::kIdNone;
240+ 
241+ codegen::Tiler tiler;
242+ codegen::TPipe tpipe("tpipe", tiler);
243+ tpipe.AddTensor(load->outputs[0]);
244+ 
245+ tiler.AddAxis(z0);
246+ tiler.AddAxis(z1);
247+ tiler.AddAxis(z2);
248+ tiler.AddAxis(z3);
249+ tiler.AddSizeVar(af::SizeVar(s0));
250+ tiler.AddSizeVar(af::SizeVar(s1));
251+ tiler.AddSizeVar(af::SizeVar(s2));
252+ tiler.AddSizeVar(af::SizeVar(s3));
253+ 
254+ codegen::ApiTensor x1;
255+ x1.id = load->outputs[0].attr.mem.tensor_id;
256+ codegen::ApiTensor y1;
257+ y1.id = store->outputs[0].attr.mem.tensor_id;
258+ codegen::CallParam cp = {"p_reg", ""};
259+ auto tensor_load = load->GetName() + "_" + load->GetOpDesc()->GetOutputNameByIndex(0);
260+ MicroApiTensor tensor1(load->outputs[0], tensor_load);
261+ auto tensor_store = store->GetName() + "_" + store->GetOpDesc()->GetOutputNameByIndex(0);
262+ MicroApiTensor tensor2(store->outputs[0], tensor_store);
263+ TensorManager tensor_mng;
264+ tensor_mng.AddTensor(tensor1);
265+ tensor_mng.AddTensor(tensor2);
266+ codegen::MicroDivApiCall call("Div");
267+ EXPECT_EQ(call.Init(div_node), 0);
268+ call.AddInput(x1.id);
269+ call.AddOutput(y1.id);
270+ 
271+ std::string result;
272+ call.Generate(tensor_mng, tpipe, cp, result);
273+ EXPECT_EQ(result, std::string{"AscendC::MicroAPI::Div<float>(vreg_1, vreg_0, p_reg);\n"});
274+ 
275+ unsetenv("AUTOFUSE_FLAGS");
276+ af::pre_process::PreProcessConfig::Instance().Reset();
277+}
278+ 
150TEST(CodegenKernel, MicroDivApiCall_Load_Div_Half_Store) {279TEST(CodegenKernel, MicroDivApiCall_Load_Div_Half_Store) {
151 af::AscGraph graph("test_div_graph");280 af::AscGraph graph("test_div_graph");
152 281 
@@ -9,6 +9,7 @@
9#include "micro_div_api_call.h"9#include "micro_div_api_call.h"
10#include "micro_api_call_factory.h"10#include "micro_api_call_factory.h"
11#include "ascir_ops.h"11#include "ascir_ops.h"
12+#include "optimize/pre_process/pre_process_config.h"
12 13 
13namespace codegen {14namespace codegen {
14Status MicroDivApiCall::Generate(const codegen::TensorManager &tensor_mng, [[maybe_unused]] const TPipe &tpipe,15Status MicroDivApiCall::Generate(const codegen::TensorManager &tensor_mng, [[maybe_unused]] const TPipe &tpipe,
@@ -26,7 +27,11 @@ Status MicroDivApiCall::Generate(const codegen::TensorManager &tensor_mng, [[may
26 static_cast<int32_t>(input_dtype));27 static_cast<int32_t>(input_dtype));
27 ss << "AscendC::MicroAPI::" << this->api_name_;28 ss << "AscendC::MicroAPI::" << this->api_name_;
28 if (input_dtype == ge::DT_FLOAT) {29 if (input_dtype == ge::DT_FLOAT) {
29- ss << "<" << input_dtype_name << ", &high_precision_div_mode" << ">";30+ ss << "<" << input_dtype_name;
31+ if (!af::pre_process::PreProcessConfig::Instance().IsInImprovePrecisionBlacklist(af::ascir_op::Div::Type)) {
32+ ss << ", &high_precision_div_mode";
33+ }
34+ ss << ">";
30 }35 }
31 ss << "(";36 ss << "(";
32 for (const auto &out_arg : this->outputs_) {37 for (const auto &out_arg : this->outputs_) {
@@ -40,7 +40,7 @@ For PyTorch, AutoFuse is enabled by configuring the `ascendc` backend through `t
40| `--enable_autofuse` | TensorFlow | Controls whether automatic fusion is enabled globally. Accepts `true` or `false`; `false` is the default. Other AutoFuse control options have no effect when it is disabled. |40| `--enable_autofuse` | TensorFlow | Controls whether automatic fusion is enabled globally. Accepts `true` or `false`; `false` is the default. Other AutoFuse control options have no effect when it is disabled. |
41| `--autofuse_enable_pass` | TensorFlow | Enables specified extended fusion capabilities. Currently, `reduce` and `concat` are supported. Multiple values are separated by commas; the default is empty and extended fusion is disabled. The same value must not be configured with `--autofuse_disable_pass`. |41| `--autofuse_enable_pass` | TensorFlow | Enables specified extended fusion capabilities. Currently, `reduce` and `concat` are supported. Multiple values are separated by commas; the default is empty and extended fusion is disabled. The same value must not be configured with `--autofuse_disable_pass`. |
42| `--autofuse_disable_pass` | TensorFlow | Disables specified extended fusion capabilities. Supports `reduce` and `concat`; multiple capabilities can be disabled by separating their values with commas. The default is empty. The same value must not be configured with `--autofuse_enable_pass`. |42| `--autofuse_disable_pass` | TensorFlow | Disables specified extended fusion capabilities. Supports `reduce` and `concat`; multiple capabilities can be disabled by separating their values with commas. The default is empty. The same value must not be configured with `--autofuse_enable_pass`. |
43-| `--autofuse_enhance_precision_blacklist` | TensorFlow | Controls whether specified AscIR operator types skip precision enhancement. Accepts comma-separated AscIR operator type strings or `all`; default: empty. `Sum`, `Mean`, and `Prod` still require precision enhancement. |43+| `--autofuse_enhance_precision_blacklist` | TensorFlow, PyTorch | Controls whether specified AscIR operator types skip precision enhancement. Accepts comma-separated AscIR operator type strings or `all`; default: empty. `Sum`, `Mean`, and `Prod` still require precision enhancement. When `Div` or `all` is configured, `Div` operators are generated with the default precision mode instead of the high-precision mode, which may affect accuracy. |
44| `--recomputation_threshold` | TensorFlow | Sets the automatic-fusion recomputation threshold. Accepts an integer from `0` to `255`; default: `1`. none. |44| `--recomputation_threshold` | TensorFlow | Sets the automatic-fusion recomputation threshold. Accepts an integer from `0` to `255`; default: `1`. none. |
45| `--max_fusion_size` | TensorFlow | Sets the maximum number of nodes in a fused operator. Accepts `0` to the maximum `uint64_t` value; `0` disables fusion; the default is implementation-defined. none. |45| `--max_fusion_size` | TensorFlow | Sets the maximum number of nodes in a fused operator. Accepts `0` to the maximum `uint64_t` value; `0` disables fusion; the default is implementation-defined. none. |
46| `--autofuse_enable_pgo` | TensorFlow, PyTorch | Enables PGO tuning by selecting better-performing Tiling through pre-run sampling. Accepts `true` or `false`; default: `false`. static graphs only, `mspti` is required, and the first configuration cannot be used with other Profiling features. |46| `--autofuse_enable_pgo` | TensorFlow, PyTorch | Enables PGO tuning by selecting better-performing Tiling through pre-run sampling. Accepts `true` or `false`; default: `false`. static graphs only, `mspti` is required, and the first configuration cannot be used with other Profiling features. |
@@ -44,7 +44,7 @@ export AUTOFUSE_FLAGS="--enable_autofuse=true"
44| `--enable_autofuse` | TensorFlow | 控制整体自动融合功能是否开启。取值为 `true` 或 `false`,`false` 为默认值;未开启时,其他 AutoFuse 控制项均不生效。 |44| `--enable_autofuse` | TensorFlow | 控制整体自动融合功能是否开启。取值为 `true` 或 `false`,`false` 为默认值;未开启时,其他 AutoFuse 控制项均不生效。 |
45| `--autofuse_enable_pass` | TensorFlow | 控制指定的扩展融合能力是否开启。目前支持 `reduce` 和 `concat`;多个取值使用英文逗号分隔,默认不配置,扩展融合默认关闭。不能与 `--autofuse_disable_pass` 配置相同取值。 |45| `--autofuse_enable_pass` | TensorFlow | 控制指定的扩展融合能力是否开启。目前支持 `reduce` 和 `concat`;多个取值使用英文逗号分隔,默认不配置,扩展融合默认关闭。不能与 `--autofuse_disable_pass` 配置相同取值。 |
46| `--autofuse_disable_pass` | TensorFlow | 控制指定的扩展融合能力是否关闭。支持配置 `reduce`、`concat`,也可以使用英文逗号分隔,同时关闭多个扩展融合能力。默认不配置;不能与 `--autofuse_enable_pass` 配置相同取值。 |46| `--autofuse_disable_pass` | TensorFlow | 控制指定的扩展融合能力是否关闭。支持配置 `reduce`、`concat`,也可以使用英文逗号分隔,同时关闭多个扩展融合能力。默认不配置;不能与 `--autofuse_enable_pass` 配置相同取值。 |
47-| `--autofuse_enhance_precision_blacklist` | TensorFlow | 控制指定 AscIR 算子类型是否跳过精度提升。取值为 AscIR 算子类型字符串,多个类型使用英文逗号分隔,也可配置为 `all`;默认值为空。`Sum`、`Mean`、`Prod` 不支持低精度类型,即使加入黑名单也仍会提升精度。 |47+| `--autofuse_enhance_precision_blacklist` | TensorFlow、PyTorch | 控制指定 AscIR 算子类型是否跳过精度提升。取值为 AscIR 算子类型字符串,多个类型使用英文逗号分隔,也可配置为 `all`;默认值为空。`Sum`、`Mean`、`Prod` 不支持低精度类型,即使加入黑名单也仍会提升精度。配置 `Div` 或 `all` 后,`Div` 算子在代码生成时关闭高精度模式,按默认精度生成,可能影响计算精度。 |
48| `--recomputation_threshold` | TensorFlow | 设置自动融合重计算阈值。取值为 `0`~`255` 的整数,默认值为 `1`。 |48| `--recomputation_threshold` | TensorFlow | 设置自动融合重计算阈值。取值为 `0`~`255` 的整数,默认值为 `1`。 |
49| `--max_fusion_size` | TensorFlow | 设置单个融合算子最多包含的节点数量。取值为 `0`~`uint64_t` 最大值,配置为 `0` 表示不融合,默认值由实现决定。 |49| `--max_fusion_size` | TensorFlow | 设置单个融合算子最多包含的节点数量。取值为 `0`~`uint64_t` 最大值,配置为 `0` 表示不融合,默认值由实现决定。 |
50| `--autofuse_enable_pgo` | TensorFlow、PyTorch | 开启 PGO 调优,通过预先上板采样选择性能更优的 Tiling。取值为 `true` 或 `false`,默认值为 `false`。仅支持静态图调优,需要准备对应版本的 `mspti`;首次配置时不能与其他 Profiling 功能同时开启。 |50| `--autofuse_enable_pgo` | TensorFlow、PyTorch | 开启 PGO 调优,通过预先上板采样选择性能更优的 Tiling。取值为 `true` 或 `false`,默认值为 `false`。仅支持静态图调优,需要准备对应版本的 `mspti`;首次配置时不能与其他 Profiling 功能同时开启。 |