已合并
feat: Add aclnn golden functions and test cases #9487
yanzhi2024创建于 8月29日
feat: Add aclnn golden functions and test cases #9487
已合并
yanzhi2024创建于 8月29日
共 48 个文件变更+803-118
@@ -12,18 +12,18 @@
12import numpy as np12import numpy as np
13 13 
14__golden__ = {14__golden__ = {
15- "kernel": {15+ "kernel": {"elu": "elu_golden"},
16- "elu": "elu_golden"
17- },
18 "aclnn": {16 "aclnn": {
19 "aclnnElu": "aclnn_elu_golden",17 "aclnnElu": "aclnn_elu_golden",
20- "aclnnInplaceElu": "aclnn_inplace_elu_golden"18+ "aclnnInplaceElu": "aclnn_inplace_elu_golden",
21- }19+ },
22}20}
23 21 
24 22 
25-def elu_golden(x, *, alpha: float = 1.0, scale: float = 1.0, input_scale: float = 1.0, **kwargs):23+def elu_golden(
26- '''24+ x, *, alpha: float = 1.0, scale: float = 1.0, input_scale: float = 1.0, **kwargs
25+):
26+ """
27 Golden function for elu.27 Golden function for elu.
28 All the parameters (names and order) follow @elu_def.cpp without outputs.28 All the parameters (names and order) follow @elu_def.cpp without outputs.
29 All the input Tensors are numpy.ndarray.29 All the input Tensors are numpy.ndarray.
@@ -34,33 +34,38 @@ def elu_golden(x, *, alpha: float = 1.0, scale: float = 1.0, input_scale: float
34 34 
35 Returns:35 Returns:
36 Output tensor36 Output tensor
37- '''37+ """
38 import torch38 import torch
39 39 
40 x_dtype = x.dtype40 x_dtype = x.dtype
41 if x_dtype.name in ("bfloat16", "float16"):41 if x_dtype.name in ("bfloat16", "float16"):
42 x = x.astype(np.float32)42 x = x.astype(np.float32)
43- 43+ 
44 x_torch = torch.from_numpy(x)44 x_torch = torch.from_numpy(x)
45- output_y = torch.ops.aten.elu(x_torch, alpha=alpha, scale=scale, input_scale=input_scale)45+ output_y = torch.ops.aten.elu(
46+ x_torch, alpha=alpha, scale=scale, input_scale=input_scale
47+ )
46 result = output_y.numpy()48 result = output_y.numpy()
47- 49+ 
48 if x_dtype.name in ("bfloat16", "float16"):50 if x_dtype.name in ("bfloat16", "float16"):
49 result = result.astype(x_dtype, copy=False)51 result = result.astype(x_dtype, copy=False)
50- 52+ 
51 return result53 return result
52 54 
53 55 
54def _aclnn_elu_impl(input_tensor, alpha, scale, inputScale):56def _aclnn_elu_impl(input_tensor, alpha, scale, inputScale):
55 import torch57 import torch
58+ 
56 alpha_val = alpha.item()59 alpha_val = alpha.item()
57 scale_val = scale.item()60 scale_val = scale.item()
58 input_scale_val = inputScale.item()61 input_scale_val = inputScale.item()
59- return torch.ops.aten.elu(input_tensor, alpha=alpha_val, scale=scale_val, input_scale=input_scale_val)62+ return torch.ops.aten.elu(
63+ input_tensor, alpha=alpha_val, scale=scale_val, input_scale=input_scale_val
64+ )
60 65 
61 66 
62-def aclnn_elu_golden(selfT, alpha, scale, inputScale, out, **kwargs):67+def aclnn_elu_golden(self, alpha, scale, inputScale, out=None, **kwargs):
63- '''68+ """
64 Aclnn golden for aclnnElu.69 Aclnn golden for aclnnElu.
65 All the parameters (name & order) follow \70 All the parameters (name & order) follow \
66 function `aclnnEluGetWorkspaceSize` in @aclnn_elu.h \71 function `aclnnEluGetWorkspaceSize` in @aclnn_elu.h \
@@ -74,12 +79,12 @@ def aclnn_elu_golden(selfT, alpha, scale, inputScale, out, **kwargs):
74 79 
75 Returns:80 Returns:
76 Output tensors.81 Output tensors.
77- '''82+ """
78- return _aclnn_elu_impl(selfT, alpha, scale, inputScale)83+ return _aclnn_elu_impl(self, alpha, scale, inputScale)
79 84 
80 85 
81def aclnn_inplace_elu_golden(selfRef, alpha, scale, inputScale, **kwargs):86def aclnn_inplace_elu_golden(selfRef, alpha, scale, inputScale, **kwargs):
82- '''87+ """
83 Aclnn golden for aclnnInplaceElu.88 Aclnn golden for aclnnInplaceElu.
84 All the parameters (name & order) follow \89 All the parameters (name & order) follow \
85 function `aclnnInplaceEluGetWorkspaceSize` in @aclnn_elu.h \90 function `aclnnInplaceEluGetWorkspaceSize` in @aclnn_elu.h \
@@ -93,5 +98,5 @@ def aclnn_inplace_elu_golden(selfRef, alpha, scale, inputScale, **kwargs):
93 98 
94 Returns:99 Returns:
95 Output tensors.100 Output tensors.
96- '''101+ """
97 return _aclnn_elu_impl(selfRef, alpha, scale, inputScale)102 return _aclnn_elu_impl(selfRef, alpha, scale, inputScale)
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,scalar_dtypes,scalar_data_ranges,absolute_precision
2+Elu_float32_ND_fuzz_1,aclnnElu,"('float32', 'float32')","('ND',)","((196, 2, 1, 76, 1, 1), (196, 2, 1, 76, 1, 1))","((-1000, -10),)","(-1,)","('float32', 'float32', 'float32')","(0, 1)",0.0001
3+Elu_float16_ND_fuzz_2,aclnnElu,"('float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('float16', 'float16', 'float16')","(0, 1)",0.001
4+Elu_float32_ND_fuzz_3,aclnnElu,"('float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)","('float32', 'float32', 'float32')","(0, 1)",0.0001
5+Elu_float16_ND_fuzz_4,aclnnElu,"('float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('float16', 'float16', 'float16')","(0, 1)",0.001
6+Elu_float16_ND_fuzz_5,aclnnElu,"('float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)","('float16', 'float16', 'float16')","(0, 1)",0.001
@@ -11,8 +11,14 @@
11# ----------------------------------------------------------------------------11# ----------------------------------------------------------------------------
12 12 
13import numpy as np13import numpy as np
14+import torch
14 15 
15-__golden__ = {"kernel": {"elu_grad_v2": "elu_grad_v2_golden"}}16+__golden__ = {
17+ "aclnn": {
18+ "aclnnEluBackward": "aclnn_elu_backward_golden",
19+ },
20+ "kernel": {"elu_grad_v2": "elu_grad_v2_golden"},
21+}
16 22 
17 23 
18def elu_grad_v2_golden(24def elu_grad_v2_golden(
@@ -54,3 +60,17 @@ def elu_grad_v2_golden(
54 if grads_dtype.name in ("bfloat16", "float16"):60 if grads_dtype.name in ("bfloat16", "float16"):
55 result = result.astype(grads_dtype, copy=False)61 result = result.astype(grads_dtype, copy=False)
56 return result62 return result
63+ 
64+ 
65+def aclnn_elu_backward_golden(
66+ gradOutput, alpha, scale, inputScale, isResult, selfOrResult, gradInput, **kwargs
67+):
68+ alpha_val = alpha.item() if hasattr(alpha, "item") else alpha
69+ scale_val = scale.item() if hasattr(scale, "item") else scale
70+ input_scale_val = inputScale.item() if hasattr(inputScale, "item") else inputScale
71+ is_result = bool(isResult.item()) if hasattr(isResult, "item") else bool(isResult)
72+ return [
73+ torch.ops.aten.elu_backward(
74+ gradOutput, alpha_val, scale_val, input_scale_val, is_result, selfOrResult
75+ )
76+ ]
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,absolute_precision,attributes,scalar_dtypes,scalar_data_ranges
2+EluGradV2_float32_ND_fuzz_1,aclnnEluBackward,"('float32', 'float32','float32')","('ND',)","((196, 2, 1, 76),(196, 2, 1, 76),(196, 2, 1, 76))","((-1000, -10),)","(-1,)","('float32',)",0.0001,{'isResult':True},"('float32', 'float32','float32')","((1,2),(3,4),(0,2))"
3+EluGradV2_float16_ND_fuzz_2,aclnnEluBackward,"('float16', 'float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('float16',)",0.001,{'isResult':True},"('float32', 'float32','float32')","((0,2),(1,2),(0,2))"
4+EluGradV2_float32_ND_fuzz_3,aclnnEluBackward,"('float32', 'float32','float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)","('float32',)",0.0001,{'isResult':False},"('float32', 'float32','float32')","((1,2),(3,4),(0,2))"
5+EluGradV2_float16_ND_fuzz_4,aclnnEluBackward,"('float16', 'float16','float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192),(139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('float16',)",0.001,{'isResult':False},"('float32', 'float32','float32')","((0,2),(1,2),(0,2))"
6+EluGradV2_float16_ND_fuzz_5,aclnnEluBackward,"('float16', 'float16','float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1),(1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)","('float16',)",0.001,{'isResult':True},"('float32', 'float32','float32')","((1,2),(3,4),(0,2))"
@@ -1,5 +1,5 @@
1#!/usr/bin/env python31#!/usr/bin/env python3
2-# -*- coding: UTF-8 -*-2+# -*- coding: utf-8 -*-
3# ----------------------------------------------------------------------------3# ----------------------------------------------------------------------------
4# Copyright (c) 2026 Huawei Technologies Co., Ltd.4# Copyright (c) 2026 Huawei Technologies Co., Ltd.
5# This program is free software, you can redistribute it and/or modify it under the terms and conditions of5# This program is free software, you can redistribute it and/or modify it under the terms and conditions of
@@ -12,7 +12,10 @@
12 12 
13import numpy as np13import numpy as np
14 14 
15-__golden__ = {"kernel": {"ge_glu_v2": "ge_glu_v2_golden"}}15+__golden__ = {
16+ "kernel": {"ge_glu_v2": "ge_glu_v2_golden"},
17+ "aclnn": {"aclnnGeGlu": "aclnn_ge_glu_golden"},
18+}
16 19 
17 20 
18def do_gelu(x, approximate):21def do_gelu(x, approximate):
@@ -37,8 +40,16 @@ def process(input_x, split_dim, activateLeft, approximate):
37 gate, x = tensor_x.chunk(2, dim=split_dim)40 gate, x = tensor_x.chunk(2, dim=split_dim)
38 else:41 else:
39 x, gate = tensor_x.chunk(2, dim=split_dim)42 x, gate = tensor_x.chunk(2, dim=split_dim)
43+ 
44+ if gate.dtype == torch.half:
45+ gate = gate.to(torch.float32)
46+ 
40 y_gelu = do_gelu(gate, approximate)47 y_gelu = do_gelu(gate, approximate)
41- y = x * y_gelu48+ 
49+ if x.dtype == torch.half:
50+ y = x * y_gelu.to(torch.half)
51+ else:
52+ y = x * y_gelu
42 return y, y_gelu53 return y, y_gelu
43 54 
44 55 
@@ -57,7 +68,35 @@ def ge_glu_v2_golden(x, *, dim=-1, approximate=1, activate_left=False, **kwargs)
57 """68 """
58 dtype = x.dtype69 dtype = x.dtype
59 if dtype != np.float64:70 if dtype != np.float64:
60- x = x.astype(np.float32)71+ if dtype != np.float16 or approximate != 1:
72+ x = x.astype(np.float32)
61 73 
62 y, y_gelu = process(x, dim, activate_left, approximate)74 y, y_gelu = process(x, dim, activate_left, approximate)
63 return y.numpy().astype(dtype), y_gelu.numpy().astype(dtype)75 return y.numpy().astype(dtype), y_gelu.numpy().astype(dtype)
76+ 
77+ 
78+def aclnn_ge_glu_golden(self, dim, approximate, out=None, outGelu=None, **kwargs):
79+ """
80+ Aclnn golden for aclnnGeGlu.
81+ Parameters follow @aclnnGeGluGetWorkspaceSize without workspaceSize & executor.
82+ All the input Tensors are torch.Tensor.
83+ """
84+ import torch
85+ 
86+ if hasattr(dim, "item"):
87+ dim = dim.item()
88+ if hasattr(approximate, "item"):
89+ approximate = approximate.item()
90+ 
91+ orig_dtype = self.dtype
92+ input_x = self
93+ if orig_dtype != torch.float64:
94+ if orig_dtype != torch.float16 or approximate != 1:
95+ input_x = input_x.to(torch.float32)
96+ 
97+ x_np = input_x.numpy()
98+ y, y_gelu = process(x_np, dim, False, approximate)
99+ 
100+ y = y.to(orig_dtype)
101+ y_gelu = y_gelu.to(orig_dtype)
102+ return [y, y_gelu]
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,attributes,absolute_precision
2+Geglu_float32_ND_fuzz_1,aclnnGeGlu,"('float32', 'float32', 'float32')","('ND',)","((1, 1, 1, 1, 1, 1, 2, 2), (1, 1, 1, 1, 1, 1, 1, 2), (1, 1, 1, 1, 1, 1, 1, 2))","((-1000, -10),)","(-2,-1)","{'dim': 6, 'approximate': 1}",0.0001
3+Geglu_float16_ND_fuzz_2,aclnnGeGlu,"('float16', 'float16', 'float16')","('ND',)","((2, 1, 271, 1, 1, 1, 2), (1, 1, 271, 1, 1, 1, 2), (1, 1, 271, 1, 1, 1, 2))","((0, 0),)","(-2,-1)","{'dim': 0, 'approximate': 1}",0.001
4+Geglu_float32_ND_fuzz_3,aclnnGeGlu,"('float32', 'float32', 'float32')","('ND',)","((1, 4), (1,2), (1,2))","((-3.4e+38, 3.4e+38),)","(-2,-1)","{'dim': -1, 'approximate': 1}",0.0001
5+Geglu_float16_ND_fuzz_4,aclnnGeGlu,"('float16', 'float16', 'float16')","('ND',)","((1, 3461, 172), (1, 3461, 86), (1, 3461, 86))","((-10, -2),)","(-2,-1)","{'dim': -1, 'approximate': 1}",0.001
6+Geglu_bfloat16_ND_fuzz_7,aclnnGeGlu,"('bfloat16', 'bfloat16', 'bfloat16')","('ND',)","((4,), (2,), (2,))","((-1, 1),)","(-2,-1)","{'dim': 0, 'approximate': 1}",0.0001
@@ -7,20 +7,23 @@
7# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,7# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
8# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.8# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
9# See LICENSE in the root of the software repository for the full text of the License.9# See LICENSE in the root of the software repository for the full text of the License.
10-'''10+"""
11gelu golden function11gelu golden function
12-'''12+"""
13+ 
13import numpy as np14import numpy as np
15+import torch
14 16 
15__golden__ = {17__golden__ = {
16- "kernel": {18+ "aclnn": {
17- "gelu": "gelu_golden"19+ "aclnnGelu": "aclnn_gelu_golden",
18- }20+ },
21+ "kernel": {"gelu": "gelu_golden"},
19}22}
20 23 
21 24 
22def gelu_golden(x, approximate="tanh", **kwargs):25def gelu_golden(x, approximate="tanh", **kwargs):
23- '''26+ """
24 Golden function for gelu.27 Golden function for gelu.
25 All the parameters (names and order) follow @gelu_def.cpp without outputs.28 All the parameters (names and order) follow @gelu_def.cpp without outputs.
26 All the input Tensors are numpy.ndarray.29 All the input Tensors are numpy.ndarray.
@@ -31,16 +34,26 @@ def gelu_golden(x, approximate="tanh", **kwargs):
31 34 
32 Returns:35 Returns:
33 Output tensor36 Output tensor
34- '''37+ """
35 import torch38 import torch
36- 39+ 
37 input_dtype = x.dtype40 input_dtype = x.dtype
38- 41+ 
39 # Promote float16 and bfloat16 to float32 for computation42 # Promote float16 and bfloat16 to float32 for computation
40 if input_dtype.name == "float16" or input_dtype.name == "bfloat16":43 if input_dtype.name == "float16" or input_dtype.name == "bfloat16":
41 x = x.astype(np.float32)44 x = x.astype(np.float32)
42- 45+ 
43 m = torch.nn.GELU(approximate=approximate)46 m = torch.nn.GELU(approximate=approximate)
44 res = m(torch.from_numpy(x)).numpy()47 res = m(torch.from_numpy(x)).numpy()
45- 48+ 
46 return res.astype(input_dtype, copy=False)49 return res.astype(input_dtype, copy=False)
50+ 
51+ 
52+def aclnn_gelu_golden(self, out, **kwargs):
53+ orig_dtype = self.dtype
54+ x = self
55+ if x.dtype != torch.float64:
56+ x = x.to(torch.float32)
57+ m = torch.nn.GELU(approximate="tanh")
58+ result = m(x).to(orig_dtype)
59+ return [result]
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,absolute_precision
2+Gelu_float32_ND_fuzz_1,aclnnGelu,"('float32', 'float32')","('ND',)","((196, 2, 1, 76),(196, 2, 1, 76))","((-1000, -10),)","(-1,)","('float32',)",0.0001
3+Gelu_float16_ND_fuzz_2,aclnnGelu,"('float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('float16',)",0.001
4+Gelu_float32_ND_fuzz_3,aclnnGelu,"('float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)","('float32',)",0.0001
5+Gelu_float16_ND_fuzz_4,aclnnGelu,"('float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('float16',)",0.001
6+Gelu_float16_ND_fuzz_5,aclnnGelu,"('float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)","('float16',)",0.001
@@ -13,9 +13,10 @@ import numpy as np
13 13 
14 14 
15__golden__ = {15__golden__ = {
16- "kernel": {16+ "aclnn": {
17- "gelu_grad": "gelu_grad_golden"17+ "aclnnGeluBackward": "aclnn_gelu_backward_golden",
18- }18+ },
19+ "kernel": {"gelu_grad": "gelu_grad_golden"},
19}20}
20 21 
21_MIN_FP32 = np.float32(2 ** (-126))22_MIN_FP32 = np.float32(2 ** (-126))
@@ -83,7 +84,7 @@ def _result_grad_compute(data_x):
83 84 
84 85 
85def gelu_grad_golden(dy, x, y, **kwargs):86def gelu_grad_golden(dy, x, y, **kwargs):
86- '''87+ """
87 Golden function for gelu_grad.88 Golden function for gelu_grad.
88 All the parameters (names and order) follow @gelu_grad_def.cpp without outputs.89 All the parameters (names and order) follow @gelu_grad_def.cpp without outputs.
89 All the input Tensors are numpy.ndarray.90 All the input Tensors are numpy.ndarray.
@@ -94,19 +95,19 @@ def gelu_grad_golden(dy, x, y, **kwargs):
94 95 
95 Returns:96 Returns:
96 Output tensor97 Output tensor
97- '''98+ """
98 import torch99 import torch
99 from packaging import version100 from packaging import version
100 101 
101 input_dtype = dy.dtype102 input_dtype = dy.dtype
102 has_improve_precision = False103 has_improve_precision = False
103- 104+ 
104 if version.parse(torch.__version__) >= version.parse("1.12.0"):105 if version.parse(torch.__version__) >= version.parse("1.12.0"):
105 if input_dtype.name in ["float16", "bfloat16"]:106 if input_dtype.name in ["float16", "bfloat16"]:
106 dy = dy.astype(np.float32)107 dy = dy.astype(np.float32)
107 x = x.astype(np.float32)108 x = x.astype(np.float32)
108 y = y.astype(np.float32)109 y = y.astype(np.float32)
109- 110+ 
110 dy_torch = torch.from_numpy(dy)111 dy_torch = torch.from_numpy(dy)
111 x_torch = torch.from_numpy(x)112 x_torch = torch.from_numpy(x)
112 result = torch.ops.aten.gelu_backward(dy_torch, x_torch, approximate="tanh")113 result = torch.ops.aten.gelu_backward(dy_torch, x_torch, approximate="tanh")
@@ -127,3 +128,19 @@ def gelu_grad_golden(dy, x, y, **kwargs):
127 if has_improve_precision:128 if has_improve_precision:
128 result = result.astype(input_dtype, copy=False)129 result = result.astype(input_dtype, copy=False)
129 return result130 return result
131+ 
132+ 
133+def aclnn_gelu_backward_golden(gradOutput, self, gradInput, **kwargs):
134+ """
135+ Aclnn golden for aclnnGeluBackward.
136+ Parameters follow @aclnnGeluBackwardGetWorkspaceSize without workspaceSize & executor.
137+ All the input Tensors are torch.Tensor.
138+ """
139+ import torch
140+ 
141+ input_dy = gradOutput
142+ input_x = self
143+ 
144+ result = torch.ops.aten.gelu_backward(input_dy, input_x, approximate="tanh")
145+ 
146+ return result
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,absolute_precision
2+GeluGrad_float32_ND_fuzz_1,aclnnGeluBackward,"('float32', 'float32', 'float32')","('ND',)","((196, 2, 1, 76),(196, 2, 1, 76),(196, 2, 1, 76))","((-1000, -10),)","(-1,)","('float32',)",0.0001
3+GeluGrad_float16_ND_fuzz_2,aclnnGeluBackward,"('float16', 'float16','float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('float16',)",0.001
4+GeluGrad_float32_ND_fuzz_3,aclnnGeluBackward,"('float32', 'float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17),(1, 1, 4, 11, 48, 17))","((-10, -2),)","(-1,)","('float32',)",0.0001
5+GeluGrad_float16_ND_fuzz_4,aclnnGeluBackward,"('float16', 'float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192),(139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('float16',)",0.001
6+GeluGrad_float16_ND_fuzz_5,aclnnGeluBackward,"('float16', 'float16','float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1),(1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)","('float16',)",0.001
@@ -12,7 +12,12 @@
12 12 
13import numpy as np13import numpy as np
14 14 
15-__golden__ = {"kernel": {"hardtanh_grad": "hardtanh_grad_golden"}}15+__golden__ = {
16+ "aclnn": {
17+ "aclnnHardtanhBackward": "aclnn_hardtanh_backward_golden",
18+ },
19+ "kernel": {"hardtanh_grad": "hardtanh_grad_golden"},
20+}
16 21 
17 22 
18def hardtanh_grad_golden(result, grad, *, min_val=-1.0, max_val=1.0, **kwargs):23def hardtanh_grad_golden(result, grad, *, min_val=-1.0, max_val=1.0, **kwargs):
@@ -38,3 +43,12 @@ def hardtanh_grad_golden(result, grad, *, min_val=-1.0, max_val=1.0, **kwargs):
38 if str(in_data_type) == "float16" or str(in_data_type) == "bfloat16":43 if str(in_data_type) == "float16" or str(in_data_type) == "bfloat16":
39 res = res.astype(in_data_type, copy=False)44 res = res.astype(in_data_type, copy=False)
40 return res45 return res
46+ 
47+ 
48+def aclnn_hardtanh_backward_golden(gradOutput, self, min, max, out, **kwargs):
49+ if hasattr(min, "item"):
50+ min = min.item()
51+ if hasattr(max, "item"):
52+ max = max.item()
53+ mask = (self > min) & (self < max)
54+ return [gradOutput * mask.to(gradOutput.dtype)]
@@ -0,0 +1,4 @@
1+testcase_name,api_name,tensor_view_shapes,tensor_dtypes,scalar_dtypes,scalar_data_ranges
2+aclnnHardtanhBackward_1,aclnnHardtanhBackward,"((1,2,3),(1,2,3),(1,2,3))","('bfloat16','bfloat16','bfloat16')","('float16','float16')","((-1, -1),(2, 2))"
3+aclnnHardtanhBackward_2,aclnnHardtanhBackward,"((1,2,3),(1,2,3),(1,2,3))","('float16','float16','float16')","('float16','float16')","((-2, -2),(3, 3))"
4+aclnnHardtanhBackward_3,aclnnHardtanhBackward,"((1,2,3),(1,2,3),(1,2,3))","('float32','float32','float32')","('float16','float16')","((10, 10),(20, 20))"
@@ -11,11 +11,17 @@
11 11 
12import numpy as np12import numpy as np
13 13 
14-__golden__ = {"kernel": {"leaky_relu": "leaky_relu_golden"}}14+__golden__ = {
15+ "aclnn": {
16+ "aclnnLeakyRelu": "aclnn_leaky_relu_golden",
17+ "aclnnInplaceLeakyRelu": "aclnn_inplace_leaky_relu_golden",
18+ },
19+ "kernel": {"leaky_relu": "leaky_relu_golden"},
20+}
15 21 
16 22 
17def leaky_relu_golden(x, *, negative_slope=0, **kwargs):23def leaky_relu_golden(x, *, negative_slope=0, **kwargs):
18- '''24+ """
19 Golden function for leaky_relu.25 Golden function for leaky_relu.
20 All the parameters (names and order) follow @leaky_relu_def.cpp without outputs.26 All the parameters (names and order) follow @leaky_relu_def.cpp without outputs.
21 All the input Tensors are numpy.ndarray.27 All the input Tensors are numpy.ndarray.
@@ -28,7 +34,7 @@ def leaky_relu_golden(x, *, negative_slope=0, **kwargs):
28 34 
29 Returns:35 Returns:
30 Output tensor36 Output tensor
31- '''37+ """
32 import torch38 import torch
33 39 
34 if "bfloat16" in x.dtype.name:40 if "bfloat16" in x.dtype.name:
@@ -40,3 +46,29 @@ def leaky_relu_golden(x, *, negative_slope=0, **kwargs):
40 if "bfloat16" in x.dtype.name:46 if "bfloat16" in x.dtype.name:
41 return result.view(torch.int16).numpy().view(x.dtype)47 return result.view(torch.int16).numpy().view(x.dtype)
42 return result.numpy()48 return result.numpy()
49+ 
50+ 
51+def aclnn_inplace_leaky_relu_golden(selfRef, negativeSlope, **kwargs):
52+ """
53+ Aclnn golden for aclnnInplaceLeakyRelu.
54+ Parameters follow @aclnnInplaceLeakyReluGetWorkspaceSize without workspaceSize & executor.
55+ All the input Tensors are torch.Tensor.
56+ """
57+ import torch
58+ 
59+ if hasattr(negativeSlope, "item"):
60+ negativeSlope = negativeSlope.item()
61+ return [torch.nn.functional.leaky_relu(selfRef, negative_slope=negativeSlope)]
62+ 
63+ 
64+def aclnn_leaky_relu_golden(self, negativeSlope, out=None, **kwargs):
65+ """
66+ Aclnn golden for aclnnLeakyRelu.
67+ Parameters follow @aclnnLeakyReluGetWorkspaceSize without workspaceSize & executor.
68+ All the input Tensors are torch.Tensor.
69+ """
70+ import torch
71+ 
72+ if hasattr(negativeSlope, "item"):
73+ negativeSlope = negativeSlope.item()
74+ return [torch.nn.functional.leaky_relu(self, negative_slope=negativeSlope)]
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,scalar_dtypes,scalar_data_ranges,attributes,absolute_precision
2+InplaceLeakyRelu_float32_ND_fuzz_1,aclnnInplaceLeakyRelu,('float32'),"('ND',)","((196, 2, 1, 76, 1, 1))","((-1000, -10),)","(-1,)","('float32',)",('float16'),"(0, 1)",{'negative_slope': 0.0006944149602797177},0.0001
3+InplaceLeakyRelu_float16_ND_fuzz_2,aclnnInplaceLeakyRelu,('float16'),"('ND',)","((192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('float16',)",('float16'),"(0, 1)",{'negative_slope': 0.007353729707176569},0.001
4+InplaceLeakyRelu_float32_ND_fuzz_3,aclnnInplaceLeakyRelu,('float32'),"('ND',)","((1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)","('float32',)",('float16'),"(0, 1)",{'negative_slope': 0.003578863762855549},0.0001
5+InplaceLeakyRelu_float16_ND_fuzz_4,aclnnInplaceLeakyRelu,('float16'),"('ND',)","((139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('float16',)",('float16'),"(0, 1)",{'negative_slope': 0.001118463241933789},0.001
6+InplaceLeakyRelu_float16_ND_fuzz_5,aclnnInplaceLeakyRelu,('float16'),"('ND',)","((1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)","('float16',)",('float16'),"(0, 1)",{'negative_slope': 0.006011051318991595},0.001
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,scalar_dtypes,scalar_data_ranges,absolute_precision
2+LeakyRelu_float32_ND_fuzz_1,aclnnLeakyRelu,"('float32', 'float32')","('ND',)","((196, 2, 1, 76, 1, 1), (196, 2, 1, 76, 1, 1))","((-1000, -10),)","(-1,)","('float32',)",('float16'),"(0, 1)",0.0001
3+LeakyRelu_float16_ND_fuzz_2,aclnnLeakyRelu,"('float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('float16',)",('float16'),"(0, 1)",0.001
4+LeakyRelu_float32_ND_fuzz_3,aclnnLeakyRelu,"('float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)","('float32',)",('float16'),"(0, 1)",0.0001
5+LeakyRelu_float16_ND_fuzz_4,aclnnLeakyRelu,"('float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('float16',)",('float16'),"(0, 1)",0.001
6+LeakyRelu_float16_ND_fuzz_5,aclnnLeakyRelu,"('float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)","('float16',)",('float16'),"(0, 1)",0.001
@@ -11,12 +11,18 @@
11# ----------------------------------------------------------------------------11# ----------------------------------------------------------------------------
12 12 
13import numpy as np13import numpy as np
14+import torch
14 15 
15-__golden__ = {"kernel": {"relu": "relu_golden"}}16+__golden__ = {
17+ "aclnn": {
18+ "aclnnRelu": "aclnn_relu_golden",
19+ },
20+ "kernel": {"relu": "relu_golden"},
21+}
16 22 
17 23 
18def relu_golden(x, **kwargs):24def relu_golden(x, **kwargs):
19- '''25+ """
20 Golden function for relu.26 Golden function for relu.
21 All the parameters (names and order) follow @relu_def.cpp without outputs.27 All the parameters (names and order) follow @relu_def.cpp without outputs.
22 All the input Tensors are numpy.ndarray.28 All the input Tensors are numpy.ndarray.
@@ -27,7 +33,16 @@ def relu_golden(x, **kwargs):
27 33 
28 Returns:34 Returns:
29 Output tensor35 Output tensor
30- '''36+ """
31 data_res = np.maximum(x, 0)37 data_res = np.maximum(x, 0)
32 data_res[np.logical_and(data_res == 0, np.signbit(data_res))] = 0.038 data_res[np.logical_and(data_res == 0, np.signbit(data_res))] = 0.0
33 return data_res39 return data_res
40+ 
41+ 
42+def aclnn_relu_golden(self, out=None, **kwargs):
43+ """
44+ Aclnn golden for aclnnRelu.
45+ Parameters follow @aclnnReluGetWorkspaceSize without workspaceSize & executor.
46+ All the input Tensors are torch.Tensor.
47+ """
48+ return [torch.relu(self)]
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,absolute_precision
2+Relu_float32_ND_fuzz_1,aclnnRelu,"('float32', 'float32')","('ND',)","((196, 2, 1, 76, 1, 1), (196, 2, 1, 76, 1, 1))","((-1000, -10),)","(-1,)","('float32',)",0.0001
3+Relu_float16_ND_fuzz_2,aclnnRelu,"('float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('float16',)",0.001
4+Relu_float32_ND_fuzz_3,aclnnRelu,"('float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)","('float32',)",0.0001
5+Relu_float16_ND_fuzz_4,aclnnRelu,"('float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('float16',)",0.001
6+Relu_float16_ND_fuzz_5,aclnnRelu,"('float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)","('float16',)",0.001
@@ -11,12 +11,19 @@
11# ----------------------------------------------------------------------------11# ----------------------------------------------------------------------------
12 12 
13import numpy as np13import numpy as np
14+import torch
14 15 
15-__golden__ = {"kernel": {"sigmoid": "sigmoid_golden"}}16+__golden__ = {
17+ "aclnn": {
18+ "aclnnSigmoid": "aclnn_sigmoid_golden",
19+ "aclnnInplaceSigmoid": "aclnn_inplace_sigmoid_golden",
20+ },
21+ "kernel": {"sigmoid": "sigmoid_golden"},
22+}
16 23 
17 24 
18def sigmoid_golden(x, **kwargs):25def sigmoid_golden(x, **kwargs):
19- '''26+ """
20 Golden function for sigmoid.27 Golden function for sigmoid.
21 All the parameters (names and order) follow @sigmoid_def.cpp without outputs.28 All the parameters (names and order) follow @sigmoid_def.cpp without outputs.
22 All the input Tensors are numpy.ndarray.29 All the input Tensors are numpy.ndarray.
@@ -27,7 +34,7 @@ def sigmoid_golden(x, **kwargs):
27 34 
28 Returns:35 Returns:
29 Output tensor36 Output tensor
30- '''37+ """
31 input_dtype = x.dtype38 input_dtype = x.dtype
32 if input_dtype.name in ("bfloat16",):39 if input_dtype.name in ("bfloat16",):
33 x = x.astype("float32")40 x = x.astype("float32")
@@ -36,3 +43,25 @@ def sigmoid_golden(x, **kwargs):
36 tensor_add = tensor_exp + 143 tensor_add = tensor_exp + 1
37 res = 1 / tensor_add44 res = 1 / tensor_add
38 return res.astype(input_dtype, copy=False)45 return res.astype(input_dtype, copy=False)
46+ 
47+ 
48+def aclnn_inplace_sigmoid_golden(selfRef=None, **kwargs):
49+ """
50+ Aclnn golden for aclnnInplaceSigmoid.
51+ Parameters follow @aclnnInplaceSigmoidGetWorkspaceSize without workspaceSize & executor.
52+ All the input Tensors are torch.Tensor.
53+ """
54+ return [
55+ torch.nn.functional.sigmoid(selfRef)
56+ if hasattr(torch.nn.functional, "sigmoid")
57+ else torch.sigmoid(selfRef)
58+ ]
59+ 
60+ 
61+def aclnn_sigmoid_golden(self, out=None, **kwargs):
62+ """
63+ Aclnn golden for aclnnSigmoid.
64+ Parameters follow @aclnnSigmoidGetWorkspaceSize without workspaceSize & executor.
65+ All the input Tensors are torch.Tensor.
66+ """
67+ return [torch.sigmoid(self)]
@@ -0,0 +1,4 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,absolute_precision
2+aclnn_sigmoid_fuzz_1,aclnnSigmoid,"('float32', 'float32')","('ND',)","((19, 2, 5, 8), (19, 2, 5, 8))","((-1000, 1000),)","(-1,)","('float32',)",0.0001
3+aclnn_sigmoid_fuzz_2,aclnnSigmoid,"('float16', 'float16')","('ND',)","((19, 2, 5, 8), (19, 2, 5, 8))","((-1000, 1000),)","(-1,)","('float16',)",0.0001
4+aclnn_sigmoid_fuzz_6,aclnnSigmoid,"('bfloat16', 'bfloat16')","('ND',)","((19, 2, 5, 8), (19, 2, 5, 8))","((-1000, 1000),)","(-1,)","('bfloat16',)",0.0001
@@ -11,12 +11,18 @@
11# ----------------------------------------------------------------------------11# ----------------------------------------------------------------------------
12 12 
13import numpy as np13import numpy as np
14+import torch
14 15 
15-__golden__ = {"kernel": {"sigmoid_grad": "sigmoid_grad_golden"}}16+__golden__ = {
17+ "aclnn": {
18+ "aclnnSigmoidBackward": "aclnn_sigmoid_backward_golden",
19+ },
20+ "kernel": {"sigmoid_grad": "sigmoid_grad_golden"},
21+}
16 22 
17 23 
18def sigmoid_grad_golden(y, dy, **kwargs):24def sigmoid_grad_golden(y, dy, **kwargs):
19- '''25+ """
20 Golden function for sigmoid_grad.26 Golden function for sigmoid_grad.
21 All the parameters (names and order) follow @sigmoid_grad_def.cpp without outputs.27 All the parameters (names and order) follow @sigmoid_grad_def.cpp without outputs.
22 All the input Tensors are numpy.ndarray.28 All the input Tensors are numpy.ndarray.
@@ -27,15 +33,24 @@ def sigmoid_grad_golden(y, dy, **kwargs):
27 33 
28 Returns:34 Returns:
29 Output tensor35 Output tensor
30- '''36+ """
31 dtype = y.dtype37 dtype = y.dtype
32- 38+ 
33- if 'float16' in str(dtype):39+ if "float16" in str(dtype):
34 y = y.astype("float32")40 y = y.astype("float32")
35 dy = dy.astype("float32")41 dy = dy.astype("float32")
36- 42+ 
37 tensor_sub = np.subtract(1.0, y)43 tensor_sub = np.subtract(1.0, y)
38 tensor_mul = np.multiply(tensor_sub, dy)44 tensor_mul = np.multiply(tensor_sub, dy)
39 res = np.multiply(tensor_mul, y)45 res = np.multiply(tensor_mul, y)
40- 46+ 
41 return res.astype(dtype, copy=False)47 return res.astype(dtype, copy=False)
48+ 
49+ 
50+def aclnn_sigmoid_backward_golden(gradOutput, output, gradInput=None, **kwargs):
51+ orig_dtype = gradOutput.dtype
52+ if orig_dtype in (torch.float16, torch.bfloat16):
53+ gradOutput = gradOutput.to(torch.float32)
54+ output = output.to(torch.float32)
55+ result = gradOutput * output * (1 - output)
56+ return [result.to(orig_dtype)]
@@ -0,0 +1,4 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,absolute_precision
2+aclnn_sigmoid_backward_fuzz_1,aclnnSigmoidBackward,"('float32', 'float32', 'float32')","('ND',)","((19, 2, 5, 8), (19, 2, 5, 8), (19, 2, 5, 8))","((-1000, 1000),)","(-1,)","('float32',)",0.0001
3+aclnn_sigmoid_backward_fuzz_2,aclnnSigmoidBackward,"('float16', 'float16', 'float16')","('ND',)","((19, 2, 5, 8), (19, 2, 5, 8), (19, 2, 5, 8))","((-1000, 1000),)","(-1,)","('float16',)",0.0001
4+aclnn_sigmoid_backward_fuzz_3,aclnnSigmoidBackward,"('bfloat16', 'bfloat16', 'bfloat16')","('ND',)","((19, 2, 5, 8), (19, 2, 5, 8), (19, 2, 5, 8))","((-1000, 1000),)","(-1,)","('bfloat16',)",0.0001
@@ -12,11 +12,16 @@
12 12 
13import numpy as np13import numpy as np
14 14 
15-__golden__ = {"kernel": {"silu_grad": "silu_grad_golden"}}15+__golden__ = {
16+ "aclnn": {
17+ "aclnnSiluBackward": "aclnn_silu_backward_golden",
18+ },
19+ "kernel": {"silu_grad": "silu_grad_golden"},
20+}
16 21 
17 22 
18def silu_grad_golden(dy, x, **kwargs):23def silu_grad_golden(dy, x, **kwargs):
19- '''24+ """
20 Golden function for silu_grad.25 Golden function for silu_grad.
21 All the parameters (names and order) follow @silu_grad_def.cpp without outputs.26 All the parameters (names and order) follow @silu_grad_def.cpp without outputs.
22 All the input Tensors are numpy.ndarray.27 All the input Tensors are numpy.ndarray.
@@ -27,11 +32,11 @@ def silu_grad_golden(dy, x, **kwargs):
27 32 
28 Returns:33 Returns:
29 Output tensor34 Output tensor
30- '''35+ """
31 import torch36 import torch
32 37 
33 def _to_torch(arr):38 def _to_torch(arr):
34- if arr.dtype.name == 'bfloat16':39+ if arr.dtype.name == "bfloat16":
35 return torch.from_numpy(arr.view(np.int16)).view(torch.bfloat16)40 return torch.from_numpy(arr.view(np.int16)).view(torch.bfloat16)
36 return torch.from_numpy(arr)41 return torch.from_numpy(arr)
37 42 
@@ -40,6 +45,17 @@ def silu_grad_golden(dy, x, **kwargs):
40 dx = torch.ops.aten.silu_backward(dy_torch, x_torch)45 dx = torch.ops.aten.silu_backward(dy_torch, x_torch)
41 46 
42 if dx.dtype == torch.bfloat16:47 if dx.dtype == torch.bfloat16:
43- return dx.view(torch.int16).numpy().view(np.dtype('bfloat16'))48+ return dx.view(torch.int16).numpy().view(np.dtype("bfloat16"))
44 else:49 else:
45 return dx.numpy()50 return dx.numpy()
51+ 
52+ 
53+def aclnn_silu_backward_golden(gradOutput, self, gradInput=None, **kwargs):
54+ """
55+ Aclnn golden for aclnnSiluBackward.
56+ Parameters follow @aclnnSiluBackwardGetWorkspaceSize without workspaceSize & executor.
57+ All the input Tensors are torch.Tensor.
58+ """
59+ import torch
60+ 
61+ return [torch.ops.aten.silu_backward(gradOutput, self)]
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,absolute_precision
2+SiluBackward_float32_ND_fuzz_1,aclnnSiluBackward,"('float32', 'float32', 'float32')","('ND',)","((196, 2, 1, 76, 1, 1), (196, 2, 1, 76, 1, 1), (196, 2, 1, 76, 1, 1))","((-1000, -10),)","(-1,)",0.0001
3+SiluBackward_float16_ND_fuzz_2,aclnnSiluBackward,"('float16', 'float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)",0.001
4+SiluBackward_float32_ND_fuzz_3,aclnnSiluBackward,"('float32', 'float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)",0.0001
5+SiluBackward_float16_ND_fuzz_4,aclnnSiluBackward,"('float16', 'float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)",0.001
6+SiluBackward_float16_ND_fuzz_5,aclnnSiluBackward,"('float16', 'float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)",0.001
@@ -11,12 +11,18 @@
11# ----------------------------------------------------------------------------11# ----------------------------------------------------------------------------
12 12 
13import numpy as np13import numpy as np
14+import torch
14 15 
15-__golden__ = {"kernel": {"swish": "swish_golden"}}16+__golden__ = {
17+ "aclnn": {
18+ "aclnnSwish": "aclnn_swish_golden",
19+ },
20+ "kernel": {"swish": "swish_golden"},
21+}
16 22 
17 23 
18def swish_golden(x, *, scale=1.0, **kwargs):24def swish_golden(x, *, scale=1.0, **kwargs):
19- '''25+ """
20 Golden function for swish.26 Golden function for swish.
21 All the parameters (names and order) follow @swish_def.cpp without outputs.27 All the parameters (names and order) follow @swish_def.cpp without outputs.
22 All the input Tensors are numpy.ndarray.28 All the input Tensors are numpy.ndarray.
@@ -27,13 +33,13 @@ def swish_golden(x, *, scale=1.0, **kwargs):
27 33 
28 Returns:34 Returns:
29 Output tensor35 Output tensor
30- '''36+ """
31 import torch37 import torch
32- 38+ 
33 dtype = x.dtype39 dtype = x.dtype
34- if dtype.name in ('float16', 'bfloat16'):40+ if dtype.name in ("float16", "bfloat16"):
35 x = x.astype(np.float32)41 x = x.astype(np.float32)
36- 42+ 
37 if scale == 1.0:43 if scale == 1.0:
38 x_torch = torch.from_numpy(x)44 x_torch = torch.from_numpy(x)
39 m = torch.nn.SiLU()45 m = torch.nn.SiLU()
@@ -44,11 +50,11 @@ def swish_golden(x, *, scale=1.0, **kwargs):
44 50 
45 51 
46def _swish_overflow(data_input, scale, dtype, **kwargs):52def _swish_overflow(data_input, scale, dtype, **kwargs):
47- short_soc_version = kwargs.get('short_soc_version', '')53+ short_soc_version = kwargs.get("short_soc_version", "")
48- 54+ 
49- if dtype.name in ('float16', 'bfloat16'):55+ if dtype.name in ("float16", "bfloat16"):
50 data_input = data_input.astype(np.float32)56 data_input = data_input.astype(np.float32)
51- 57+ 
52 if short_soc_version in ("Ascend950",):58 if short_soc_version in ("Ascend950",):
53 scale_arr = np.array([scale], dtype=data_input.dtype)59 scale_arr = np.array([scale], dtype=data_input.dtype)
54 multi = data_input * scale_arr * -1.060 multi = data_input * scale_arr * -1.0
@@ -59,15 +65,25 @@ def _swish_overflow(data_input, scale, dtype, **kwargs):
59 scale_arr = np.array([scale], dtype=data_input.dtype)65 scale_arr = np.array([scale], dtype=data_input.dtype)
60 scale_input = np.multiply(data_input, scale_arr)66 scale_input = np.multiply(data_input, scale_arr)
61 abs_scale_input = np.abs(scale_input)67 abs_scale_input = np.abs(scale_input)
62- minus_abs = np.multiply(abs_scale_input, np.array([-1.0], dtype=data_input.dtype))68+ minus_abs = np.multiply(
69+ abs_scale_input, np.array([-1.0], dtype=data_input.dtype)
70+ )
63 sign_diff = np.add(scale_input, minus_abs)71 sign_diff = np.add(scale_input, minus_abs)
64 half_sign_diff = np.multiply(sign_diff, np.array([0.5], dtype=data_input.dtype))72 half_sign_diff = np.multiply(sign_diff, np.array([0.5], dtype=data_input.dtype))
65- 73+ 
66 exp_top = np.exp(half_sign_diff)74 exp_top = np.exp(half_sign_diff)
67 exp_bottom = np.exp(minus_abs)75 exp_bottom = np.exp(minus_abs)
68 one_plus_exp = np.add(exp_bottom, np.array([1.0], dtype=data_input.dtype))76 one_plus_exp = np.add(exp_bottom, np.array([1.0], dtype=data_input.dtype))
69- 77+ 
70 input_mul_exp = np.multiply(data_input, exp_top)78 input_mul_exp = np.multiply(data_input, exp_top)
71 res = np.divide(input_mul_exp, one_plus_exp)79 res = np.divide(input_mul_exp, one_plus_exp)
72- 80+ 
73 return res.astype(dtype, copy=False)81 return res.astype(dtype, copy=False)
82+ 
83+ 
84+def aclnn_swish_golden(self, betaOptional=0, out=None, **kwargs):
85+ if hasattr(betaOptional, "item"):
86+ betaOptional = betaOptional.item()
87+ if betaOptional == 1.0 or betaOptional == 0:
88+ return [torch.nn.functional.silu(self)]
89+ return [self * torch.sigmoid(self) ** betaOptional]
@@ -0,0 +1,4 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,scalar_dtypes,scalar_data_ranges,absolute_precision
2+aclnn_swish_fuzz_1,aclnnSwish,"('float32', 'float32')","('ND',)","((19, 2, 5, 8), (19, 2, 5, 8))","((-1000, 1000),)","(-1,)","('float32',)","('float32',)","(1, 1)",0.0001
3+aclnn_swish_fuzz_2,aclnnSwish,"('float16', 'float16')","('ND',)","((19, 2, 5, 8), (19, 2, 5, 8))","((-1000, 1000),)","(-1,)","('float16',)","('float32',)","(1, 1)",0.0001
4+aclnn_swish_fuzz_6,aclnnSwish,"('bfloat16', 'bfloat16')","('ND',)","((19, 2, 5, 8), (19, 2, 5, 8))","((-1000, 1000),)","(-1,)","('bfloat16',)","('float32',)","(1, 1)",0.0001
@@ -13,7 +13,12 @@ import torch
13from ttk.utilities.dtypes import numpy_to_torch_tensor, torch_to_numpy_tensor13from ttk.utilities.dtypes import numpy_to_torch_tensor, torch_to_numpy_tensor
14 14 
15 15 
16-__golden__ = {"kernel": {"threshold_grad_v2_d": "threshold_grad_v2_d_golden"}}16+__golden__ = {
17+ "aclnn": {
18+ "aclnnThresholdBackward": "aclnn_threshold_backward_golden",
19+ },
20+ "kernel": {"threshold_grad_v2_d": "threshold_grad_v2_d_golden"},
21+}
17 22 
18 23 
19def threshold_grad_v2_d_golden(grad_output, self_tensor, *, threshold=1.0, **kwargs):24def threshold_grad_v2_d_golden(grad_output, self_tensor, *, threshold=1.0, **kwargs):
@@ -32,3 +37,10 @@ def threshold_grad_v2_d_golden(grad_output, self_tensor, *, threshold=1.0, **kwa
32 mask = self_t.to(torch.float32) > float(threshold)37 mask = self_t.to(torch.float32) > float(threshold)
33 result = torch.where(mask, grad_output_t, torch.zeros_like(grad_output_t))38 result = torch.where(mask, grad_output_t, torch.zeros_like(grad_output_t))
34 return torch_to_numpy_tensor(result.cpu())39 return torch_to_numpy_tensor(result.cpu())
40+ 
41+ 
42+def aclnn_threshold_backward_golden(gradOutput, self, threshold, out, **kwargs):
43+ if hasattr(threshold, "item"):
44+ threshold = threshold.item()
45+ mask = (self > threshold).to(gradOutput.dtype)
46+ return [gradOutput * mask]
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,scalar_dtypes,scalar_data_ranges,absolute_precision
2+aclnn_relugrad_fuzz_1,aclnnThresholdBackward,"('float32', 'float32', 'float32')","('ND',)","((19, 2, 5, 8), (19, 2, 5, 8), (19, 2, 5, 8))","((-1000, 1000),)","(-1,)","('float32',)",('float32'),"(0, 0)",0.0001
3+aclnn_relugrad_fuzz_2,aclnnThresholdBackward,"('float16', 'float16', 'float16')","('ND',)","((19, 2, 5, 8), (19, 2, 5, 8), (19, 2, 5, 8))","((-1000, 1000),)","(-1,)","('float16',)",('float32'),"(0, 0)",0.0001
4+aclnn_relugrad_fuzz_3,aclnnThresholdBackward,"('int32', 'int32', 'int32')","('ND',)","((19, 2, 5, 8), (19, 2, 5, 8), (19, 2, 5, 8))","((-1000, 1000),)","(-1,)","('int32',)",('float32'),"(0, 0)",0.0001
5+aclnn_relugrad_fuzz_4,aclnnThresholdBackward,"('int8', 'int8', 'int8')","('ND',)","((19, 2, 5, 8), (19, 2, 5, 8), (19, 2, 5, 8))","((-1000, 1000),)","(-1,)","('int8',)",('float32'),"(0, 0)",0.0001
6+aclnn_relugrad_fuzz_5,aclnnThresholdBackward,"('uint8', 'uint8', 'uint8')","('ND',)","((19, 2, 5, 8), (19, 2, 5, 8), (19, 2, 5, 8))","((-1000, 1000),)","(-1,)","('uint8',)",('float32'),"(0, 0)",0.0001
@@ -11,7 +11,14 @@
11# ----------------------------------------------------------------------------11# ----------------------------------------------------------------------------
12 12 
13 13 
14-__golden__ = {"kernel": {"bucketize_v2": "bucketize_v2_golden"}}14+import torch
15+ 
16+__golden__ = {
17+ "aclnn": {
18+ "aclnnBucketize": "aclnn_bucketize_golden",
19+ },
20+ "kernel": {"bucketize_v2": "bucketize_v2_golden"},
21+}
15 22 
16 23 
17def bucketize_v2_golden(x, boundaries, out_int32=False, right=False, **kwargs):24def bucketize_v2_golden(x, boundaries, out_int32=False, right=False, **kwargs):
@@ -29,3 +36,15 @@ def bucketize_v2_golden(x, boundaries, out_int32=False, right=False, **kwargs):
29 boundaries_t = torch.from_numpy(boundaries)36 boundaries_t = torch.from_numpy(boundaries)
30 res = torch.bucketize(data_t, boundaries_t, out_int32=out_int32, right=right)37 res = torch.bucketize(data_t, boundaries_t, out_int32=out_int32, right=right)
31 return res.numpy()38 return res.numpy()
39+ 
40+ 
41+def aclnn_bucketize_golden(self, boundaries, outInt32=0, right=0, out=None, **kwargs):
42+ if hasattr(outInt32, "item"):
43+ outInt32 = bool(outInt32.item())
44+ elif isinstance(outInt32, int):
45+ outInt32 = bool(outInt32)
46+ if hasattr(right, "item"):
47+ right = bool(right.item())
48+ elif isinstance(right, int):
49+ right = bool(right)
50+ return [torch.bucketize(self, boundaries, out_int32=outInt32, right=right)]
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,output_tensor_indexes,input_data_ranges,attributes
2+aclnnBucketize_int8_bf16_ND_infnan_random_000000,aclnnBucketize,"['int8', 'bf16', 'int64']","('ND',)","[[3, 9, 4, 1, 5, 4, 6, 4], [31628], [3, 9, 4, 1, 5, 4, 6, 4]]","(-1,)","[[inf, inf], [-5375, 67165]]","{'outInt32':False , 'right':False}"
3+aclnnBucketize_fp32_fp16_ND_infnan_random_000002,aclnnBucketize,"['fp32', 'fp16', 'int32']","('ND',)","[[9875], [13917], [9875]]","(-1,)","[[-inf, -inf], [-68700, 36819]]","{'outInt32':True , 'right':False}"
4+aclnnBucketize_int8_int64_ND_infnan_random_000003,aclnnBucketize,"['int8', 'int64', 'int32']","('ND',)","[[3, 5, 2, 610, 7, 3], [9262], [3, 5, 2, 610, 7, 3]]","(-1,)","[[nan, nan], [-81938, 50073]]","{'outInt32':True , 'right':False}"
5+aclnnBucketize_int64_ND_infnan_random_000004,aclnnBucketize,"['int64', 'int64', 'int32']","('ND',)","[[5, 2, 5, 5, 7], [14382], [5, 2, 5, 5, 7]]","(-1,)","[[0, 1], [-92999, 94508]]","{'outInt32':True , 'right':False}"
6+aclnnBucketize_int8_fp64_ND_infnan_random_000006,aclnnBucketize,"['int8', 'fp64', 'int64']","('ND',)","[[2, 91, 2, 2888, 2], [36828], [2, 91, 2, 2888, 2]]","(-1,)","[[-inf, -inf], [-96569, 84815]]","{'outInt32':False , 'right':True}"
@@ -11,7 +11,12 @@
11# ----------------------------------------------------------------------------11# ----------------------------------------------------------------------------
12 12 
13 13 
14-__golden__ = {"kernel": {"embedding": "embedding_golden"}}14+__golden__ = {
15+ "aclnn": {
16+ "aclnnEmbedding": "aclnn_embedding_golden",
17+ },
18+ "kernel": {"embedding": "embedding_golden"},
19+}
15 20 
16 21 
17def embedding_golden(x, indices, **kwargs):22def embedding_golden(x, indices, **kwargs):
@@ -53,3 +58,14 @@ def embedding_golden(x, indices, **kwargs):
53 )58 )
54 59 
55 return res60 return res
61+ 
62+ 
63+def aclnn_embedding_golden(weight, indices, out=None, **kwargs):
64+ """
65+ Aclnn golden for aclnnEmbedding.
66+ Parameters follow @aclnnEmbeddingGetWorkspaceSize without workspaceSize & executor.
67+ All the input Tensors are torch.Tensor.
68+ """
69+ import torch
70+ 
71+ return torch.nn.functional.embedding(indices, weight)
@@ -0,0 +1,6 @@
1+testcase_name,network_name,api_name,tensor_view_shapes,tensor_formats,tensor_dtypes,tensor_storage_shapes,tensor_view_offsets,tensor_view_strides,output_tensor_indexes,output_inplace_indexes,attributes,scalar_dtypes,input_data_ranges,precision_tolerances,absolute_precision,scalar_data_ranges,is_enabled
2+sdxl_aclnn_func_case_212,UNKNOWN,aclnnEmbedding,"((49408, 768), (1, 77), (1, 77, 768))","('ND', 'ND', 'ND')","('bfloat16', 'int64', 'bfloat16')","((37945344,), (77,), (59136,))","(0, 0, 0)","((768, 1), (77, 1), (59136, 768, 1))","(2,)",(),{},(),"((None, None), (0, 49407))",,1.00E-08,"((None, None),)",TRUE
3+sdxl_aclnn_func_case_215,UNKNOWN,aclnnEmbedding,"((77, 1280), (1, 77), (1, 77, 1280))","('ND', 'ND', 'ND')","('bfloat16', 'int64', 'bfloat16')","((98560,), (77,), (98560,))","(0, 0, 0)","((1280, 1), (77, 1), (98560, 1280, 1))","(2,)",(),{},(),"((None, None), (0, 76))",,1.00E-08,"((None, None),)",TRUE
4+sdxl_aclnn_func_case_213,UNKNOWN,aclnnEmbedding,"((77, 768), (1, 77), (1, 77, 768))","('ND', 'ND', 'ND')","('bfloat16', 'int64', 'bfloat16')","((59136,), (77,), (59136,))","(0, 0, 0)","((768, 1), (77, 1), (59136, 768, 1))","(2,)",(),{},(),"((None, None), (0, 76))",,1.00E-08,"((None, None),)",TRUE
5+sdxl_aclnn_func_case_214,UNKNOWN,aclnnEmbedding,"((49408, 1280), (1, 77), (1, 77, 1280))","('ND', 'ND', 'ND')","('bfloat16', 'int64', 'bfloat16')","((63242240,), (77,), (98560,))","(0, 0, 0)","((1280, 1), (77, 1), (98560, 1280, 1))","(2,)",(),{},(),"((None, None), (0, 49407))",,1.00E-08,"((None, None),)",TRUE
6+aclnn_embedding_0,UNKNOWN,aclnnEmbedding,"((3, 32), (128, 5), (128, 5, 32))","('ND', 'ND', 'ND')","('float16', 'int64', 'float16')","((96,), (640,), (20480,))","(0, 0, 0)","((32, 1), (5, 1), (160, 32, 1))","(2,)",(),{},(),"((None, None), (0, 2))",,1.00E-08,"((None, None),)",TRUE
@@ -0,0 +1,34 @@
1+#!/usr/bin/env python3
2+# -*- coding: UTF-8 -*-
3+# ----------------------------------------------------------------------------
4+# Copyright (c) 2026 Huawei Technologies Co., Ltd.
5+# This program is free software, you can redistribute it and/or modify it under the terms and conditions of
6+# CANN Open Software License Agreement Version 2.0 (the "License").
7+# Please refer to the License for details. You may not use this file except in compliance with the License.
8+# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
9+# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
10+# See LICENSE in the root of the software repository for the full text of the License.
11+# ----------------------------------------------------------------------------
12+import torch
13+ 
14+__golden__ = {
15+ "aclnn": {
16+ "aclnnGather": "aclnn_gather_golden",
17+ }
18+}
19+ 
20+ 
21+def aclnn_gather_golden(self, dim, index, out=None, **kwargs):
22+ """
23+ Aclnn golden for aclnnGather.
24+ Parameters follow @aclnnGatherGetWorkspaceSize without workspaceSize & executor.
25+ All the input Tensors are torch.Tensor.
26+ """
27+ 
28+ x = self
29+ tensor_x = x
30+ index = index
31+ dim = dim
32+ np_out = torch.gather(input=tensor_x, dim=dim, index=index)
33+ 
34+ return np_out
@@ -0,0 +1,4 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,attributes,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes
2+gather_elements_test_case_0,aclnnGather,"('bfloat16', 'int64', 'bfloat16')","('ND',)",{ 'dim': -4},"((1, 41, 1, 1, 1), (1, 35, 1, 1, 1), (1, 35, 1, 1, 1))","((0, 0.001), (0, 1))","(2,)",('bfloat16')
3+gather_elements_test_case_1,aclnnGather,"('int32', 'int64', 'int32')","('ND',)",{ 'dim': 0},"((1, 1, 1, 1, 1, 127, 1, 1), (1, 1, 1, 1, 1, 127, 1, 1), (1, 1, 1, 1, 1, 127, 1, 1))","((-1147483648, 2147483647), (0, 0))","(2,)",('int32')
4+gather_elements_test_case_2,aclnnGather,"('float32', 'int64', 'float32')","('ND',)",{ 'dim': 0},"((1, 1, 11, 1, 1, 1), (1, 1, 11, 1, 1, 1), (1, 1, 11, 1, 1, 1))","((2, 10), (0, 0))","(2,)",('float32')
@@ -10,17 +10,20 @@
10 10 
11 11 
12import numpy as np12import numpy as np
13+import torch
13 14 
14 15 
15__golden__ = {16__golden__ = {
16- "kernel": {17+ "aclnn": {
17- "index_fill_d": "index_fill_d_golden"18+ "aclnnInplaceIndexFillTensor": "aclnn_inplace_index_fill_tensor_golden",
18- }19+ "aclnnIndexFillTensor": "aclnn_index_fill_tensor_golden",
20+ },
21+ "kernel": {"index_fill_d": "index_fill_d_golden"},
19}22}
20 23 
21 24 
22def index_fill_d_golden(x, assist1, assist2, **kwargs):25def index_fill_d_golden(x, assist1, assist2, **kwargs):
23- '''26+ """
24 Golden function for index_fill_d.27 Golden function for index_fill_d.
25 All the parameters (names and order) follow @index_fill_d_def.cpp without outputs.28 All the parameters (names and order) follow @index_fill_d_def.cpp without outputs.
26 All the input Tensors are numpy.ndarray.29 All the input Tensors are numpy.ndarray.
@@ -31,6 +34,30 @@ def index_fill_d_golden(x, assist1, assist2, **kwargs):
31 34 
32 Returns:35 Returns:
33 Output tensor36 Output tensor
34- '''37+ """
35 output_y = np.where(assist1 > 0, x, assist2)38 output_y = np.where(assist1 > 0, x, assist2)
36 return output_y39 return output_y
40+ 
41+ 
42+def aclnn_index_fill_tensor_golden(self, dim, index, value, out=None, **kwargs):
43+ if hasattr(dim, "item"):
44+ dim = dim.item()
45+ if hasattr(value, "item"):
46+ value = value.item()
47+ if not isinstance(index, torch.Tensor):
48+ index = torch.tensor(index)
49+ result = self.clone()
50+ result.index_fill_(dim, index.long(), value)
51+ return [result]
52+ 
53+ 
54+def aclnn_inplace_index_fill_tensor_golden(selfRef, dim, index, value, **kwargs):
55+ if hasattr(dim, "item"):
56+ dim = dim.item()
57+ if hasattr(value, "item"):
58+ value = value.item()
59+ if not isinstance(index, torch.Tensor):
60+ index = torch.tensor(index)
61+ result = selfRef.clone()
62+ result.index_fill_(dim, index.long(), value)
63+ return [result]
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,attributes,tensor_view_shapes,scalar_dtypes,scalar_data_ranges,output_tensor_indexes,input_data_ranges,precision_tolerances,strict_precision_mode,is_enabled
2+aclnnIndexFillTensor_fp16_N_000001,aclnnIndexFillTensor,"('float32','float32')","('ND',)","{'dim':-2,'index':[1, 2]}","((7,52),(7,52))","('float32', )","((79,79),)","(1,)","((-2,-1))","((0.001, 0.001),)",,1
3+aclnnIndexFillTensor_fp16_N_000002,aclnnIndexFillTensor,"('float16','float16')","('ND',)","{'dim':0,'index':[0]}","((4,8,16),(4,8,16))","('float16', )","((10,10),)","(1,)","((0,1))","((0.001, 0.001),)",,1
4+aclnnIndexFillTensor_int32_N_000003,aclnnIndexFillTensor,"('int32','int32')","('ND',)","{'dim':1,'index':[0,1,2]}","((5,10,20),(5,10,20))","('int32', )","((100,100),)","(1,)","((-100,100))","((0.001, 0.001),)",,1
5+aclnnIndexFillTensor_int64_N_000004,aclnnIndexFillTensor,"('int64','int64')","('ND',)","{'dim':-1,'index':[3]}","((8,16,32),(8,16,32))","('int64', )","((50,50),)","(1,)","((-50,50))","((0.001, 0.001),)",,1
6+aclnnIndexFillTensor_fp16_N_000005,aclnnIndexFillTensor,"('float16','float16')","('ND',)","{'dim':0,'index':[0,2,4]}","((6,12,24),(6,12,24))","('float16', )","((-5,-5),)","(1,)","((-1,1))","((0.001, 0.001),)",,1
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,attributes,tensor_view_shapes,output_inplace_indexes,scalar_dtypes,scalar_data_ranges,input_data_ranges,precision_tolerances,is_enabled
2+aclnnInplaceIndexFillTensor_fp16_N_000001,aclnnInplaceIndexFillTensor,"('float32',)","('ND',)","{'dim':-2,'index':[1, 2]}","((7,52),)","(0,)","('float32', )","((79,79),)","((-2,-1))","((0.001, 0.001),)",1
3+aclnnInplaceIndexFillTensor_fp16_N_000002,aclnnInplaceIndexFillTensor,"('float16',)","('ND',)","{'dim':0,'index':[0]}","((4,8,16),)","(0,)","('float16', )","((10,10),)","((0,1))","((0.001, 0.001),)",1
4+aclnnInplaceIndexFillTensor_int32_N_000003,aclnnInplaceIndexFillTensor,"('int32',)","('ND',)","{'dim':1,'index':[0,1,2]}","((5,10,20),)","(0,)","('int32', )","((100,100),)","((-100,100))","((0.001, 0.001),)",1
5+aclnnInplaceIndexFillTensor_int64_N_000004,aclnnInplaceIndexFillTensor,"('int64',)","('ND',)","{'dim':-1,'index':[3]}","((8,16,32),)","(0,)","('int64', )","((50,50),)","((-50,50))","((0.001, 0.001),)",1
6+aclnnInplaceIndexFillTensor_fp16_N_000005,aclnnInplaceIndexFillTensor,"('float16',)","('ND',)","{'dim':0,'index':[0,2,4]}","((6,12,24),)","(0,)","('float16', )","((-5,-5),)","((-1,1))","((0.001, 0.001),)",1
@@ -10,13 +10,18 @@
10# See LICENSE in the root of the software repository for the full text of the License.10# See LICENSE in the root of the software repository for the full text of the License.
11# ----------------------------------------------------------------------------11# ----------------------------------------------------------------------------
12 12 
13-import numpy as np13+import torch
14 14 
15-__golden__ = {"kernel": {"masked_scatter": "masked_scatter_golden"}}15+__golden__ = {
16+ "aclnn": {
17+ "aclnnInplaceMaskedScatter": "aclnn_inplace_masked_scatter_golden",
18+ },
19+ "kernel": {"masked_scatter": "masked_scatter_golden"},
20+}
16 21 
17 22 
18def masked_scatter_golden(input0, input1, input2, **kwargs):23def masked_scatter_golden(input0, input1, input2, **kwargs):
19- '''24+ """
20 Golden function for masked_scatter.25 Golden function for masked_scatter.
21 All the parameters (names and order) follow @masked_scatter_def.cpp without outputs.26 All the parameters (names and order) follow @masked_scatter_def.cpp without outputs.
22 All the input Tensors are numpy.ndarray.27 All the input Tensors are numpy.ndarray.
@@ -27,8 +32,7 @@ def masked_scatter_golden(input0, input1, input2, **kwargs):
27 32 
28 Returns:33 Returns:
29 Output tensor34 Output tensor
30- '''35+ """
31- import torch
32 36 
33 dtype = input0.dtype37 dtype = input0.dtype
34 if "bfloat16" in str(dtype):38 if "bfloat16" in str(dtype):
@@ -44,3 +48,9 @@ def masked_scatter_golden(input0, input1, input2, **kwargs):
44 if "bfloat16" in str(dtype):48 if "bfloat16" in str(dtype):
45 res = res.view(dtype)49 res = res.view(dtype)
46 return res50 return res
51+ 
52+ 
53+def aclnn_inplace_masked_scatter_golden(selfRef, mask, source, **kwargs):
54+ result = selfRef.clone()
55+ result.masked_scatter_(mask.bool(), source)
56+ return [result]
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_view_shapes,tensor_dtypes,output_tensor_indexes,precision_tolerances,input_data_ranges,is_enabled
2+aclnnMaskedScatter_001,aclnnInplaceMaskedScatter,"((3, 4), (3, 4), (3, 4))","('float', 'bool', 'float')","(0,)",,,
3+aclnnMaskedScatter_002,aclnnInplaceMaskedScatter,"((3, 4, 5), (3, 4, 5), (3, 4, 5))","('float16', 'bool', 'float16')","(0,)",,,
4+aclnnMaskedScatter_003,aclnnInplaceMaskedScatter,"((3, 4), (3, 4), (3, 4))","('double', 'bool', 'double')","(0,)",,,
5+aclnnMaskedScatter_004,aclnnInplaceMaskedScatter,"((3, 4), (3, 4), (3, 4))","('uint8', 'bool', 'uint8')","(0,)",,,
6+aclnnMaskedScatter_005,aclnnInplaceMaskedScatter,"((3, 4), (3, 4), (3, 4))","('int8', 'bool', 'int8')","(0,)",,,
@@ -11,12 +11,18 @@
11# ----------------------------------------------------------------------------11# ----------------------------------------------------------------------------
12 12 
13import numpy as np13import numpy as np
14+import torch
14 15 
15-__golden__ = {"kernel": {"scatter": "scatter_golden"}}16+__golden__ = {
17+ "aclnn": {
18+ "aclnnInplaceScatterUpdate": "aclnn_inplace_scatter_update_golden",
19+ },
20+ "kernel": {"scatter": "scatter_golden"},
21+}
16 22 
17 23 
18def scatter_golden(var, indices, update_value, *, axis=0, **kwargs):24def scatter_golden(var, indices, update_value, *, axis=0, **kwargs):
19- '''25+ """
20 Golden function for scatter.26 Golden function for scatter.
21 All the parameters (names and order) follow @scatter_def.cpp without outputs.27 All the parameters (names and order) follow @scatter_def.cpp without outputs.
22 All the input Tensors are numpy.ndarray.28 All the input Tensors are numpy.ndarray.
@@ -27,32 +33,43 @@ def scatter_golden(var, indices, update_value, *, axis=0, **kwargs):
27 33 
28 Returns:34 Returns:
29 Output tensor35 Output tensor
30- '''36+ """
31 import copy37 import copy
32- 38+ 
33- dtype_dict = {"float32": 4, "int8": 1, "int32": 4, "float16": 2, "int64": 8, "bfloat16": 2}39+ dtype_dict = {
40+ "float32": 4,
41+ "int8": 1,
42+ "int32": 4,
43+ "float16": 2,
44+ "int64": 8,
45+ "bfloat16": 2,
46+ }
34 all_shape = len(var.shape)47 all_shape = len(var.shape)
35 abs_axis = axis48 abs_axis = axis
36 if axis < 0:49 if axis < 0:
37 abs_axis = all_shape + axis50 abs_axis = all_shape + axis
38- 51+ 
39- if not (all_shape == 4 and abs_axis == 3 and var.shape[2] % dtype_dict.get(str(var.dtype), 1) == 052+ if not (
40- and var.shape[3] % dtype_dict.get(str(var.dtype), 1) == 0):53+ all_shape == 4
54+ and abs_axis == 3
55+ and var.shape[2] % dtype_dict.get(str(var.dtype), 1) == 0
56+ and var.shape[3] % dtype_dict.get(str(var.dtype), 1) == 0
57+ ):
41 trans_shape_0 = var.shape[0]58 trans_shape_0 = var.shape[0]
42 update_shape_0 = update_value.shape[0]59 update_shape_0 = update_value.shape[0]
43- 60+ 
44 seceond_dim = 161 seceond_dim = 1
45 update_second_dim = 162 update_second_dim = 1
46- 63+ 
47 for i in range(1, abs_axis):64 for i in range(1, abs_axis):
48 seceond_dim *= var.shape[i]65 seceond_dim *= var.shape[i]
49 update_second_dim *= update_value.shape[i]66 update_second_dim *= update_value.shape[i]
50 trans_shape_1 = seceond_dim67 trans_shape_1 = seceond_dim
51 update_shape_1 = update_second_dim68 update_shape_1 = update_second_dim
52- 69+ 
53 trans_shape_2 = var.shape[abs_axis]70 trans_shape_2 = var.shape[abs_axis]
54 update_shape_2 = update_value.shape[abs_axis]71 update_shape_2 = update_value.shape[abs_axis]
55- 72+ 
56 fourth_dim = 173 fourth_dim = 1
57 update_fourth_dim = 174 update_fourth_dim = 1
58 for i in range(abs_axis + 1, all_shape):75 for i in range(abs_axis + 1, all_shape):
@@ -60,27 +77,33 @@ def scatter_golden(var, indices, update_value, *, axis=0, **kwargs):
60 update_fourth_dim *= update_value.shape[i]77 update_fourth_dim *= update_value.shape[i]
61 trans_shape_3 = fourth_dim78 trans_shape_3 = fourth_dim
62 update_shape_3 = update_fourth_dim79 update_shape_3 = update_fourth_dim
63- 80+ 
64 var = var.reshape(trans_shape_0, trans_shape_1, trans_shape_2, trans_shape_3)81 var = var.reshape(trans_shape_0, trans_shape_1, trans_shape_2, trans_shape_3)
65- update_value = update_value.reshape(update_shape_0, update_shape_1, update_shape_2, update_shape_3)82+ update_value = update_value.reshape(
66- 83+ update_shape_0, update_shape_1, update_shape_2, update_shape_3
84+ )
85+ 
67 axis = 286 axis = 2
68- 87+ 
69 shape_0 = update_value.shape[0]88 shape_0 = update_value.shape[0]
70 shape_2 = update_value.shape[2]89 shape_2 = update_value.shape[2]
71 shape_3 = update_value.shape[3]90 shape_3 = update_value.shape[3]
72 output = copy.deepcopy(var)91 output = copy.deepcopy(var)
73 indices_value = indices.astype(np.int64)92 indices_value = indices.astype(np.int64)
74- 93+ 
75 if len(indices.shape) == 2:94 if len(indices.shape) == 2:
76 if axis == -2 or axis == 2:95 if axis == -2 or axis == 2:
77 for i in range(indices.shape[0]):96 for i in range(indices.shape[0]):
78 for k in range(shape_2):97 for k in range(shape_2):
79- output[indices_value[i][0], :, indices_value[i][1] + k, :] = update_value[i, :, k, :]98+ output[indices_value[i][0], :, indices_value[i][1] + k, :] = (
99+ update_value[i, :, k, :]
100+ )
80 elif axis == -1 or axis == 3:101 elif axis == -1 or axis == 3:
81 for i in range(indices.shape[0]):102 for i in range(indices.shape[0]):
82- for l in range(shape_3):103+ for idx_l in range(shape_3):
83- output[indices_value[i][0], :, :, indices_value[i][1] + l] = update_value[i, :, :, l]104+ output[indices_value[i][0], :, :, indices_value[i][1] + idx_l] = (
105+ update_value[i, :, :, idx_l]
106+ )
84 else:107 else:
85 if axis == -2 or axis == 2:108 if axis == -2 or axis == 2:
86 for i in range(shape_0):109 for i in range(shape_0):
@@ -90,7 +113,28 @@ def scatter_golden(var, indices, update_value, *, axis=0, **kwargs):
90 elif axis == -1 or axis == 3:113 elif axis == -1 or axis == 3:
91 for i in range(shape_0):114 for i in range(shape_0):
92 indices_key = indices_value[i]115 indices_key = indices_value[i]
93- for l in range(shape_3):116+ for idx_l in range(shape_3):
94- output[i, :, :, indices_key + l] = update_value[i, :, :, l]117+ output[i, :, :, indices_key + idx_l] = update_value[i, :, :, idx_l]
95- 118+ 
96 return output119 return output
120+ 
121+ 
122+def aclnn_inplace_scatter_update_golden(data, indices, updates, axis, **kwargs):
123+ if hasattr(axis, "item"):
124+ axis = axis.item()
125+ 
126+ orig_dtype = data.dtype
127+ 
128+ if orig_dtype == torch.bfloat16:
129+ data_np = data.to(torch.float32).numpy()
130+ updates_np = updates.to(torch.float32).numpy()
131+ else:
132+ data_np = data.numpy()
133+ updates_np = updates.numpy()
134+ 
135+ indices_np = indices.to(torch.int64).numpy()
136+ 
137+ result_np = scatter_golden(data_np, indices_np, updates_np, axis=axis)
138+ 
139+ result = torch.from_numpy(result_np).to(orig_dtype)
140+ return [result]
@@ -0,0 +1,5 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,attributes,tensor_view_shapes,input_data_ranges,output_dtypes,output_tensor_indexes
2+aclnnInplaceScatterUpdate_test_case_0,aclnnInplaceScatterUpdate,"('float32', 'int64', 'float32')","('ND',)",{ 'axis': -5},"((4, 3, 2, 5, 6, 5, 6, 5), (4,), (4, 3, 2, 2, 6, 5, 6, 5))","((-1, -0.01), (0, 3), (-2, -1))",('float32'),"(0,)"
3+aclnnInplaceScatterUpdate_test_case_1,aclnnInplaceScatterUpdate,"('float16', 'int32', 'float16')","('ND',)",{ 'axis': -4},"((1, 1, 1, 1, 1), (1,), (1, 1, 1, 1, 1))","((-65504.0, -65504.0), (0, 0), (-1, -0.01))",('float16'),"(0,)"
4+aclnnInplaceScatterUpdate_test_case_2,aclnnInplaceScatterUpdate,"('float16', 'int32', 'float16')","('ND',)",{ 'axis': -1},"((1, 1), (1, 2), (1, 1))","((-2, -1), (0, 0), (-2, -1))",('float16'),"(0,)"
5+aclnnInplaceScatterUpdate_test_case_3,aclnnInplaceScatterUpdate,"('bfloat16', 'int64', 'bfloat16')","('ND',)",{ 'axis': 5},"((7, 7, 8, 8, 7, 9, 9, 7), (1, 2), (1, 7, 8, 8, 7, 9, 9, 7))","((-10, -2), (0, 0), (-1000, -10))",('bfloat16'),"(0,)"
@@ -11,9 +11,15 @@
11# ----------------------------------------------------------------------------11# ----------------------------------------------------------------------------
12 12 
13import numpy as np13import numpy as np
14+import torch
14 15 
15 16 
16-__golden__ = {"kernel": {"adaptive_max_pool3d": "adaptive_max_pool3d_golden"}}17+__golden__ = {
18+ "aclnn": {
19+ "aclnnAdaptiveMaxPool3d": "aclnn_adaptive_max_pool3d_golden",
20+ },
21+ "kernel": {"adaptive_max_pool3d": "adaptive_max_pool3d_golden"},
22+}
17 23 
18 24 
19def adaptive_max_pool3d_golden(x, output_size, indices_dtype=3, **kwargs):25def adaptive_max_pool3d_golden(x, output_size, indices_dtype=3, **kwargs):
@@ -41,3 +47,16 @@ def adaptive_max_pool3d_golden(x, output_size, indices_dtype=3, **kwargs):
41 indices = indices.to(torch.int32).numpy() ## 与竞品差异47 indices = indices.to(torch.int32).numpy() ## 与竞品差异
42 48 
43 return output, indices49 return output, indices
50+ 
51+ 
52+def aclnn_adaptive_max_pool3d_golden(
53+ self, outputSize=0, outputOut=None, indicesOut=None, **kwargs
54+):
55+ if hasattr(outputSize, "tolist"):
56+ outputSize = outputSize.tolist()
57+ elif isinstance(outputSize, int):
58+ outputSize = [outputSize] * 3
59+ result = torch.nn.functional.adaptive_max_pool3d(
60+ self, outputSize, return_indices=True
61+ )
62+ return [result[0], result[1]]
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,attributes,tensor_view_shapes,scalar_dtypes,scalar_data_ranges,output_tensor_indexes,input_data_ranges,precision_tolerances,is_enabled
2+aclnn_adaptive_max_pool3d_test_00001,aclnnAdaptiveMaxPool3d,"('bfloat16', 'bfloat16', 'int32')","('ND',)","{'outputSize': [57, 8, 37]}","([0, 100, 71, 82], [0, 57, 8, 37], [0, 57, 8, 37])",(),"((None, None),)","(1, 2)","[[-100, 100]]",,TRUE
3+aclnn_adaptive_max_pool3d_test_00002,aclnnAdaptiveMaxPool3d,"('bfloat16', 'bfloat16', 'int64')","('NCDHW',)","{'outputSize': [6, 13, 19]}","([0, 19, 19, 21, 23], [0, 19, 6, 13, 19], [0, 19, 6, 13, 19])",(),"((None, None),)","(1, 2)","[[-100, 100]]",,TRUE
4+aclnn_adaptive_max_pool3d_test_00003,aclnnAdaptiveMaxPool3d,"('float32', 'float32', 'int32')","('ND',)","{'outputSize': [17, 3, 15]}","([47, 32, 6, 22], [47, 17, 3, 15], [47, 17, 3, 15])",(),"((None, None),)","(1, 2)","[[inf, inf]]",,TRUE
5+aclnn_adaptive_max_pool3d_test_00004,aclnnAdaptiveMaxPool3d,"('float16', 'float16', 'int64')","('ND',)","{'outputSize': [2, 148, 704]}","([5, 5, 153, 8191], [5, 2, 148, 704], [5, 2, 148, 704])",(),"((None, None),)","(1, 2)","[[-inf, inf]]",,TRUE
6+aclnn_adaptive_max_pool3d_test_00005,aclnnAdaptiveMaxPool3d,"('float32', 'float32', 'int32')","('ND',)","{'outputSize': [116, 198, 2]}","([2, 298, 352, 2], [2, 116, 198, 2], [2, 116, 198, 2])",(),"((None, None),)","(1, 2)","[[-inf, -inf]]",,TRUE
@@ -11,8 +11,14 @@
11# ----------------------------------------------------------------------------11# ----------------------------------------------------------------------------
12 12 
13import numpy as np13import numpy as np
14+import torch
14 15 
15-__golden__ = {"kernel": {"ascend_quant_v2": "ascend_quant_v2_golden"}}16+__golden__ = {
17+ "aclnn": {
18+ "aclnnAscendQuantV3": "aclnn_ascend_quant_v3_golden",
19+ },
20+ "kernel": {"ascend_quant_v2": "ascend_quant_v2_golden"},
21+}
16 22 
17 23 
18def ascend_quant_v2_golden(24def ascend_quant_v2_golden(
@@ -83,3 +89,28 @@ def ascend_quant_v2_golden(
83 round_data = round_data.astype(hifloat8, copy=False)89 round_data = round_data.astype(hifloat8, copy=False)
84 90 
85 return round_data91 return round_data
92+ 
93+ 
94+def aclnn_ascend_quant_v3_golden(
95+ x, scale, offset, sqrtMode, roundMode, dstType, axis, y, **kwargs
96+):
97+ if hasattr(sqrtMode, "item"):
98+ sqrtMode = bool(sqrtMode.item())
99+ if hasattr(roundMode, "decode"):
100+ roundMode = roundMode.decode()
101+ x_f = x.to(torch.float32)
102+ scale_f = scale.to(torch.float32)
103+ offset_f = offset.to(torch.float32) if offset is not None else None
104+ scale_rst = x_f * (scale_f**2) if sqrtMode else x_f * scale_f
105+ if offset_f is not None:
106+ scale_rst = scale_rst + offset_f
107+ if roundMode == "round":
108+ round_data = torch.round(scale_rst)
109+ elif roundMode == "floor":
110+ round_data = torch.floor(scale_rst)
111+ elif roundMode == "ceil":
112+ round_data = torch.ceil(scale_rst)
113+ else:
114+ round_data = torch.round(scale_rst)
115+ round_data = round_data.clamp(-128, 127).to(torch.int8)
116+ return [round_data]
@@ -0,0 +1,2 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,attributes,input_data_ranges,output_tensor_indexes,absolute_precision
2+aclnnAscendQuantV3_float16_ND_fuzz_1,aclnnAscendQuantV3,"('float16', 'float16', 'float16', 'int8')","('ND',)","((16, 1024),(1024,),(1024,),(16, 1024))","{'dstType': 2, 'sqrtMode':False, 'roundMode':'round', 'axis':1}","((-2, 2), (-2, 2), (-2, 2))","(-1,)",0.0001
@@ -11,8 +11,15 @@
11# ----------------------------------------------------------------------------11# ----------------------------------------------------------------------------
12 12 
13import numpy as np13import numpy as np
14+import torch
14 15 
15-__golden__ = {"kernel": {"dynamic_quant": "dynamic_quant_golden"}}16+__golden__ = {
17+ "aclnn": {
18+ "aclnnDynamicQuantV3": "aclnn_dynamic_quant_v3_golden",
19+ "aclnnDynamicQuant": "aclnn_dynamic_quant_golden",
20+ },
21+ "kernel": {"dynamic_quant": "dynamic_quant_golden"},
22+}
16 23 
17 24 
18def _dynamic_quant_common(25def _dynamic_quant_common(
@@ -247,3 +254,82 @@ def dynamic_quant_golden(
247 is_symmetrical,254 is_symmetrical,
248 output_dtype_str,255 output_dtype_str,
249 )256 )
257+ 
258+ 
259+def aclnn_dynamic_quant_golden(x, smoothScalesOptional, yOut, scaleOut, **kwargs):
260+ """
261+ Aclnn golden for aclnnDynamicQuant.
262+ """
263+ x_f = x.to(torch.float32) if x.dtype != torch.float32 else x
264+ smooth_scales = (
265+ smoothScalesOptional.to(torch.float32)
266+ if smoothScalesOptional is not None
267+ else None
268+ )
269+ x_scaled = x_f * smooth_scales if smooth_scales is not None else x_f
270+ amax = torch.amax(
271+ torch.abs(x_scaled).view(-1, x_scaled.shape[-1]), dim=-1, keepdim=True
272+ )
273+ scale = amax / 127.0
274+ scale = torch.where(scale == 0, torch.ones_like(scale), scale)
275+ quantized = torch.round(x_scaled / scale)
276+ quantized = quantized.clamp(-128, 127).to(torch.int8)
277+ return [quantized, scale]
278+ 
279+ 
280+def aclnn_dynamic_quant_v3_golden(
281+ x,
282+ smoothScalesOptional,
283+ groupIndexOptional,
284+ dstType,
285+ isSymmetrical,
286+ quantMode,
287+ yOut,
288+ scaleOut,
289+ offsetOut,
290+ **kwargs,
291+):
292+ """
293+ Aclnn golden for aclnnDynamicQuantV3.
294+ """
295+ if hasattr(dstType, "item"):
296+ dstType = dstType.item()
297+ if hasattr(isSymmetrical, "item"):
298+ isSymmetrical = bool(isSymmetrical.item())
299+ if hasattr(quantMode, "item"):
300+ quantMode = quantMode.item()
301+ x_f = x.to(torch.float32) if x.dtype != torch.float32 else x
302+ smooth_scales = (
303+ smoothScalesOptional.to(torch.float32)
304+ if smoothScalesOptional is not None
305+ else None
306+ )
307+ x_scaled = x_f * smooth_scales if smooth_scales is not None else x_f
308+ scale_max = 127.0
309+ scale_max_no_sym = 255.0
310+ offset = None
311+ if not isSymmetrical:
312+ input_max = torch.max(x_scaled, dim=-1, keepdim=True).values
313+ input_min = torch.min(x_scaled, dim=-1, keepdim=True).values
314+ scale = (input_max - input_min) / scale_max_no_sym
315+ scale = torch.where(scale == 0, torch.ones_like(scale), scale)
316+ offset = scale_max - (input_max / scale)
317+ input_scaled = x_scaled / scale + offset
318+ else:
319+ input_abs = torch.abs(x_scaled)
320+ input_max = torch.max(input_abs, dim=-1, keepdim=True).values
321+ scale = input_max / scale_max
322+ scale = torch.where(scale == 0, torch.ones_like(scale), scale)
323+ input_scaled = x_scaled / scale
324+ round_data = torch.round(input_scaled)
325+ if dstType == 2:
326+ if isSymmetrical:
327+ round_data = round_data.clamp(-128, 127).to(torch.int8)
328+ else:
329+ round_data = round_data.clamp(0, 255).to(torch.uint8)
330+ scale_out = scale.squeeze(-1)
331+ if offset is not None:
332+ offset_out = offset.squeeze(-1)
333+ else:
334+ offset_out = torch.zeros_like(scale_out)
335+ return [round_data, scale_out, offset_out]
@@ -0,0 +1,2 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,absolute_precision
2+aclnnDynamicQuant_float16_ND_fuzz_1,aclnnDynamicQuant,"('float16', 'float16', 'int8', 'float32')","('ND',)","((16, 1024),(1024,),(16, 1024),(16,))","((-2, 2), (-2, 2))","(2, 3)",0.0001
@@ -11,8 +11,14 @@
11# ----------------------------------------------------------------------------11# ----------------------------------------------------------------------------
12 12 
13import numpy as np13import numpy as np
14+import torch
14 15 
15-__golden__ = {"kernel": {"quantize": "quantize_golden"}}16+__golden__ = {
17+ "aclnn": {
18+ "aclnnQuantize": "aclnn_quantize_golden",
19+ },
20+ "kernel": {"quantize": "quantize_golden"},
21+}
16 22 
17 23 
18def quantize_golden(x, scales, zero_points=None, *, dtype, axis=1, **kwargs):24def quantize_golden(x, scales, zero_points=None, *, dtype, axis=1, **kwargs):
@@ -84,3 +90,25 @@ def quantize_golden(x, scales, zero_points=None, *, dtype, axis=1, **kwargs):
84 round_data = round_data.astype(hifloat8, copy=False)90 round_data = round_data.astype(hifloat8, copy=False)
85 91 
86 return round_data92 return round_data
93+ 
94+ 
95+def aclnn_quantize_golden(x, scales, zeroPoints, dtype, axis, out, **kwargs):
96+ if hasattr(dtype, "item"):
97+ dtype = dtype.item()
98+ if hasattr(axis, "item"):
99+ axis = axis.item()
100+ x_f = x.to(torch.float32)
101+ scale = scales.to(torch.float32)
102+ offset = zeroPoints.to(torch.float32) if zeroPoints is not None else None
103+ x_shape = x_f.shape
104+ scale_shape = scale.shape
105+ if len(x_shape) != len(scale_shape):
106+ tmp_scale_shape = [1] * len(x_shape)
107+ tmp_scale_shape[axis] = scale_shape[0]
108+ scale = scale.reshape(tmp_scale_shape)
109+ scale_rst = x_f / scale
110+ if offset is not None:
111+ offset = offset.reshape(scale.shape)
112+ scale_rst = scale_rst + offset
113+ round_data = torch.round(scale_rst).clamp(-128, 127).to(torch.int8)
114+ return [round_data]
@@ -0,0 +1,3 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,attributes,absolute_precision
2+aclnn_quantize_float32_ND_fuzz_1,aclnnQuantize,"('bfloat16', 'bfloat16', 'bfloat16', 'int8')","('ND',)","((19, 2, 3), (3,), (3,), (19, 2, 3))","((-10, 10),)","(-1,)","{'dtype':2, 'axis': -1}",0.0001
3+aclnn_quantize_float32_ND_fuzz_2,aclnnQuantize,"('float32', 'float16', 'int32', 'int8')","('ND',)","((19, 2, 3), (1,), (1,), (19, 2, 3))","((-10, 10),)","(-1,)","{'dtype':2, 'axis': -1}",0.0001