已合并
feat:nn add aclnn testcase and golden again. #9634
yanzhi2024创建于 9月1日
feat:nn add aclnn testcase and golden again. #9634
已合并
yanzhi2024创建于 9月1日
共 41 个文件变更+1492-107
@@ -45,7 +45,7 @@ def sigmoid_golden(x, **kwargs):
45 return res.astype(input_dtype, copy=False)45 return res.astype(input_dtype, copy=False)
46 46 
47 47 
48-def aclnn_inplace_sigmoid_golden(selfRef=None, **kwargs):48+def aclnn_inplace_sigmoid_golden(selfRef, **kwargs):
49 """49 """
50 Aclnn golden for aclnnInplaceSigmoid.50 Aclnn golden for aclnnInplaceSigmoid.
51 Parameters follow @aclnnInplaceSigmoidGetWorkspaceSize without workspaceSize & executor.51 Parameters follow @aclnnInplaceSigmoidGetWorkspaceSize without workspaceSize & executor.
@@ -0,0 +1,4 @@
1+id,testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_inplace_indexes,absolute_precision
2+1,aclnn_inplace_sigmoid_fuzz_1,aclnnInplaceSigmoid,('float32'),"('ND',)","((19, 2, 5, 8),)","((-1000, 1000),)","(-1,)","(-1,)",0.0001
3+2,aclnn_inplace_sigmoid_fuzz_2,aclnnInplaceSigmoid,('float16'),"('ND',)","((19, 2, 5, 8),)","((-1000, 1000),)","(-1,)","(-1,)",0.0001
4+6,aclnn_inplace_sigmoid_fuzz_6,aclnnInplaceSigmoid,('bfloat16'),"('ND',)","((19, 2, 5, 8),)","((-1000, 1000),)","(-1,)","(-1,)",0.0001
@@ -11,7 +11,12 @@
11# ----------------------------------------------------------------------------11# ----------------------------------------------------------------------------
12 12 
13 13 
14-__golden__ = {"kernel": {"embedding_dense_grad_v2": "embedding_dense_grad_v2_golden"}}14+__golden__ = {
15+ "aclnn": {
16+ "aclnnEmbeddingDenseBackward": "aclnn_embedding_dense_backward_golden",
17+ },
18+ "kernel": {"embedding_dense_grad_v2": "embedding_dense_grad_v2_golden"},
19+}
15 20 
16 21 
17def embedding_dense_grad_v2_golden(22def embedding_dense_grad_v2_golden(
@@ -61,3 +66,36 @@ def embedding_dense_grad_v2_golden(
61 else:66 else:
62 result = result.numpy()67 result = result.numpy()
63 return result68 return result
69+ 
70+ 
71+def aclnn_embedding_dense_backward_golden(
72+ grad, indices, numWeights=0, paddingIdx=0, scaleGradByFreq=0, out=None, **kwargs
73+):
74+ """
75+ Aclnn golden for aclnnEmbeddingDenseBackward.
76+ Parameters follow @aclnnEmbeddingDenseBackwardGetWorkspaceSize without workspaceSize & executor.
77+ All the input Tensors are torch.Tensor.
78+ """
79+ import torch
80+ 
81+ grad_dtype = grad.dtype
82+ indices_dtype = indices.dtype
83+ 
84+ if grad_dtype == torch.float16 or grad_dtype == torch.bfloat16:
85+ grad = grad.to(torch.float32)
86+ 
87+ if indices_dtype != torch.int32 and indices_dtype != torch.int64:
88+ if indices_dtype == torch.float64:
89+ indices = indices.to(torch.int64)
90+ else:
91+ indices = indices.to(torch.int32)
92+ attrs = kwargs.get("attributes", {})
93+ num_weights = attrs.get("numWeights", -1)
94+ padding_idx = attrs.get("paddingIdx", -1)
95+ scale_grad_by_freq = attrs.get("scaleGradByFreq", False)
96+ result = torch.ops.aten.embedding_dense_backward(
97+ grad, indices, num_weights, padding_idx, scale_grad_by_freq
98+ )
99+ if grad_dtype == torch.float16 or grad_dtype == torch.bfloat16:
100+ result = result.to(grad_dtype)
101+ return result
@@ -0,0 +1,6 @@
1+testcase_name,network_name,api_name,tensor_view_shapes,tensor_formats,tensor_dtypes,tensor_storage_shapes,tensor_view_offsets,tensor_view_strides,output_tensor_indexes,output_inplace_indexes,attributes,scalar_dtypes,input_data_ranges,precision_tolerances,absolute_precision,tensor_list_distribution,scalar_list_distribution,scalar_data_ranges,is_enabled
2+aclnnEmbeddingDenseBackward_v2_fp16_int32_nd_random_01,UNKNOWN,aclnnEmbeddingDenseBackward,"((24, 430, 4), (1032, 1, 2, 5), (1000, 4))","('ND',)","('float16', 'int32', 'float16')",(),"(0,)",(),"(2,)",(),"{'numWeights': 1000, 'paddingIdx': -1, 'scaleGradByFreq': False}",(),"((-300, -300), (0, 999))","(0.005,0.005)",1.00E-08,(),(),"((None, None),)",TRUE
3+aclnnEmbeddingDenseBackward_v2_fp16_int32_nd_random_02,UNKNOWN,aclnnEmbeddingDenseBackward,"((13, 37, 39, 22), (18759,), (800, 22))","('ND',)","('float16', 'int32', 'float16')",(),"(0,)",(),"(2,)",(),"{'numWeights': 800, 'paddingIdx': 4, 'scaleGradByFreq': True}",(),"((10, 1000), (0, 799))","(0.005,0.005)",1.00E-08,(),(),"((None, None),)",TRUE
4+aclnnEmbeddingDenseBackward_v2_fp16_int64_nd_random_01,UNKNOWN,aclnnEmbeddingDenseBackward,"((1229, 968), (1, 1, 1229, 1, 1), (86, 968))","('ND',)","('float16', 'int64', 'float16')",(),"(0,)",(),"(2,)",(),"{'numWeights': 86, 'paddingIdx': 1, 'scaleGradByFreq': False}",(),"((-2, -1), (0, 85))","(0.005,0.005)",1.00E-08,(),(),"((None, None),)",TRUE
5+aclnnEmbeddingDenseBackward_v2_fp16_int64_nd_random_02,UNKNOWN,aclnnEmbeddingDenseBackward,"((2525, 76), (2525,), (81, 76))","('ND',)","('float16', 'int64', 'float16')",(),"(0,)",(),"(2,)",(),"{'numWeights': 81, 'paddingIdx': -3, 'scaleGradByFreq': False}",(),"((-10, -2), (0, 80))","(0.005,0.005)",1.00E-08,(),(),"((None, None),)",TRUE
6+aclnnEmbeddingDenseBackward_v2_fp32_int32_nd_random_01,UNKNOWN,aclnnEmbeddingDenseBackward,"((3, 61, 2), (3, 61, 1), (82, 2))","('ND',)","('float32', 'int32', 'float32')",(),"(0,)",(),"(2,)",(),"{'numWeights': 82, 'paddingIdx': 3, 'scaleGradByFreq': False}",(),"((-1000, -10), (0, 81))","(0.0005,0.0005)",1.00E-08,(),(),"((None, None),)",TRUE
@@ -10,13 +10,17 @@
10# See LICENSE in the root of the software repository for the full text of the License.10# See LICENSE in the root of the software repository for the full text of the License.
11# ----------------------------------------------------------------------------11# ----------------------------------------------------------------------------
12 12 
13-import numpy as np
14 13 
15-__golden__ = {"kernel": {"gather_nd": "gather_nd_golden"}}14+__golden__ = {
15+ "aclnn": {
16+ "aclnnGatherNd": "aclnn_gather_nd_golden",
17+ },
18+ "kernel": {"gather_nd": "gather_nd_golden"},
19+}
16 20 
17 21 
18def gather_nd_golden(params, indices, **kwargs):22def gather_nd_golden(params, indices, **kwargs):
19- '''23+ """
20 Golden function for gather_nd.24 Golden function for gather_nd.
21 All the parameters (names and order) follow @gather_nd_def.cpp without outputs.25 All the parameters (names and order) follow @gather_nd_def.cpp without outputs.
22 All the input Tensors are numpy.ndarray.26 All the input Tensors are numpy.ndarray.
@@ -27,25 +31,57 @@ def gather_nd_golden(params, indices, **kwargs):
27 31 
28 Returns:32 Returns:
29 Output tensor33 Output tensor
30- '''34+ """
31 import tensorflow as tf35 import tensorflow as tf
36+ 
32 tf.compat.v1.disable_eager_execution()37 tf.compat.v1.disable_eager_execution()
33 38 
34 params_shape = params.shape39 params_shape = params.shape
35 indices_shape = indices.shape40 indices_shape = indices.shape
36 41 
37 data_dtype = params.dtype42 data_dtype = params.dtype
38- if data_dtype.name == 'bfloat16':43+ if data_dtype.name == "bfloat16":
39- params = params.view('int16')44+ params = params.view("int16")
40 45 
41 params_ph = tf.compat.v1.placeholder(dtype=params.dtype.name, shape=params_shape)46 params_ph = tf.compat.v1.placeholder(dtype=params.dtype.name, shape=params_shape)
42 indices_ph = tf.compat.v1.placeholder(dtype=indices.dtype.name, shape=indices_shape)47 indices_ph = tf.compat.v1.placeholder(dtype=indices.dtype.name, shape=indices_shape)
43 48 
44 with tf.compat.v1.Session() as sess:49 with tf.compat.v1.Session() as sess:
45- gather_res = tf.compat.v1.gather_nd(params_ph, indices_ph, name=None, batch_dims=0)50+ gather_res = tf.compat.v1.gather_nd(
51+ params_ph, indices_ph, name=None, batch_dims=0
52+ )
46 res = sess.run(gather_res, feed_dict={params_ph: params, indices_ph: indices})53 res = sess.run(gather_res, feed_dict={params_ph: params, indices_ph: indices})
47 54 
48- if data_dtype.name == 'bfloat16':55+ if data_dtype.name == "bfloat16":
49 res = res.view(data_dtype)56 res = res.view(data_dtype)
50 57 
51 return res58 return res
59+ 
60+ 
61+def aclnn_gather_nd_golden(self, indices, negativeIndexSupport=0, out=None, **kwargs):
62+ """
63+ Aclnn golden for aclnnGatherNd.
64+ """
65+ import torch
66+ import tensorflow as tf
67+ 
68+ tf.compat.v1.disable_eager_execution()
69+ 
70+ params_data = self
71+ indices_data = indices
72+ 
73+ params_shape = params_data.shape
74+ indices_shape = indices_data.shape
75+ 
76+ params_dtype = str(params_data.dtype)[6:]
77+ indices_dtype = str(indices_data.dtype)[6:]
78+ params = tf.compat.v1.placeholder(dtype=params_dtype, shape=params_shape)
79+ indices = tf.compat.v1.placeholder(dtype=indices_dtype, shape=indices_shape)
80+ 
81+ with tf.compat.v1.Session() as sess:
82+ gather_res = tf.compat.v1.gather_nd(params, indices, name=None, batch_dims=0)
83+ res = sess.run(
84+ gather_res, feed_dict={params: params_data, indices: indices_data}
85+ )
86+ 
87+ return torch.from_numpy(res)
@@ -12,35 +12,61 @@
12 12 
13import numpy as np13import numpy as np
14 14 
15-__input__ = {"kernel": {"gather_nd": "gather_nd_input"}}15+__input__ = {
16+ "kernel": {"gather_nd": "gather_nd_input"},
17+ "aclnn": {"aclnnGatherNd": "aclnn_gather_nd_input"},
18+}
16 19 
17 20 
18def gather_nd_input(x, indices, **kwargs):21def gather_nd_input(x, indices, **kwargs):
19- '''22+ """Input function for gather_nd (kernel)."""
20- Input function for gather_nd.
21- All the parameters (names and order) follow @gather_nd_def.cpp without outputs.
22- All the input Tensors are numpy.ndarray.
23- 
24- Args:
25- **kwargs: input_dtypes, full_soc_version, short_soc_version, testcase_name
26- full_soc_version, short_soc_version, testcase_name
27- 
28- Returns:
29- Input tensors list
30- '''
31 if str(x.dtype) != "bool":23 if str(x.dtype) != "bool":
32 params = np.arange(0, x.size, 1, dtype=x.dtype).reshape(x.shape)24 params = np.arange(0, x.size, 1, dtype=x.dtype).reshape(x.shape)
33 else:25 else:
34- params = np.random.choice(a=[False, True], size=x.shape, p=[0.5, 0.5]).reshape(x.shape)26+ params = np.random.choice(a=[False, True], size=x.shape, p=[0.5, 0.5]).reshape(
27+ x.shape
28+ )
35 29 
36 ranks = indices.shape[-1]30 ranks = indices.shape[-1]
37 res_indices = []31 res_indices = []
38 for rank in range(0, ranks):32 for rank in range(0, ranks):
39- indices_rank = np.random.uniform(0, params.shape[rank], (1,)).astype(indices.dtype)33+ indices_rank = np.random.uniform(0, params.shape[rank], (1,)).astype(
34+ indices.dtype
35+ )
40 res_indices.append(indices_rank.item())36 res_indices.append(indices_rank.item())
41 37 
42 for index in indices.shape[0:-1]:38 for index in indices.shape[0:-1]:
43 res_indices = res_indices * index39 res_indices = res_indices * index
44- res_indices = np.reshape(res_indices, indices.shape).astype(indices.dtype, copy=False)40+ res_indices = np.reshape(res_indices, indices.shape).astype(
41+ indices.dtype, copy=False
42+ )
45 43 
46 return [params, res_indices]44 return [params, res_indices]
45+ 
46+ 
47+def aclnn_gather_nd_input(*args, **kwargs):
48+ """Input function for aclnnGatherNd. Generate valid indices."""
49+ import torch
50+ 
51+ params = args[0] if len(args) > 0 else kwargs.get("self")
52+ indices = args[1] if len(args) > 1 else kwargs.get("indices")
53+ 
54+ if params is None or indices is None:
55+ return list(args)
56+ 
57+ params_shape = params.shape
58+ ranks = indices.shape[-1]
59+ 
60+ res_indices = []
61+ for rank in range(ranks):
62+ max_idx = params_shape[rank] - 1
63+ idx = np.random.uniform(0, max_idx + 1, (1,)).astype(np.int64)
64+ res_indices.append(idx.item())
65+ 
66+ for index in indices.shape[0:-1]:
67+ res_indices = res_indices * index
68+ res_indices = np.reshape(res_indices, indices.shape).astype(np.int64)
69+ 
70+ # Convert to torch tensor with original dtype
71+ indices_torch = torch.from_numpy(res_indices).to(indices.dtype)
72+ indices.copy_(indices_torch)
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,attributes,tensor_view_shapes,input_data_ranges,out_put_tensor_indexes,precision_tolerances
2+testcase0,aclnnGatherNd,"('float16', 'int32', 'float16')","('ND',)",{'negativeIndexSupport': False},"((2, 2, 4, 2, 2, 1, 8), (6,), (8,))","((-1.0, 1.0), (0.0, 1))","(2,)","((0.001, 0.001),)"
3+testcase1,aclnnGatherNd,"('float32', 'int64', 'float32')","('ND',)",{'negativeIndexSupport': True},"((240, 49, 7, 11), (1,), (49, 7, 11))","((1, 2), (0, 1))","(2,)","((0.001, 0.001),)"
4+testcase2,aclnnGatherNd,"('float16', 'int64', 'float16')","('ND',)",{'negativeIndexSupport': True},"((43, 112, 80), (47, 1), (47, 112, 80))","((-0.001, 0), (0, 0))","(2,)","((0.001, 0.001),)"
5+testcase3,aclnnGatherNd,"('float32', 'int32', 'float32')","('ND',)",{'negativeIndexSupport': True},"((1, 128, 1), (1, 1, 1, 1, 1, 2), (1, 1, 1, 1, 1, 1))","((-3.4e+38, 3.4e+38), (0, 1))","(2,)","((0.001, 0.001),)"
6+testcase4,aclnnGatherNd,"('float16', 'int64', 'float16')","('ND',)",{'negativeIndexSupport': True},"((1, 1, 58, 122, 6, 90, 47, 1), (5,), (90, 47, 1))","((-0.01, -0.001), (0, 1))","(2,)","((0.001, 0.001),)"
@@ -11,7 +11,12 @@
11# ----------------------------------------------------------------------------11# ----------------------------------------------------------------------------
12 12 
13 13 
14-__golden__ = {"kernel": {"gather_v2": "gather_v2_golden"}}14+__golden__ = {
15+ "aclnn": {
16+ "aclnnGatherV2": "aclnn_gather_v2_golden",
17+ },
18+ "kernel": {"gather_v2": "gather_v2_golden"},
19+}
15 20 
16 21 
17def gather_v2_golden(22def gather_v2_golden(
@@ -66,3 +71,24 @@ def gather_v2_golden(
66 res = res.view(data_dtype)71 res = res.view(data_dtype)
67 72 
68 return res73 return res
74+ 
75+ 
76+def aclnn_gather_v2_golden(self, dim, index, out=None, **kwargs):
77+ """
78+ Aclnn golden for aclnnGatherV2.
79+ """
80+ import tensorflow as tf
81+ import torch
82+ 
83+ tensor_x = self
84+ x_dtype = self.dtype
85+ if "bfloat16" in str(x_dtype):
86+ tensor_x = self.to(torch.float32)
87+ 
88+ if hasattr(dim, "item"):
89+ dim = dim.item()
90+ 
91+ tf_out = tf.gather(tensor_x, index, axis=dim)
92+ np_out = tf_out.numpy()
93+ pt_out = torch.from_numpy(np_out)
94+ return pt_out
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,attributes,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes
2+gather_v2_test_case_0,aclnnGatherV2,"('int32', 'int32', 'int32')","('ND',)",{ 'dim': 5},"((9, 6, 5, 6, 9, 5), (1,), (9, 6, 5, 6, 9, 1))","((-1, 1), (0, 4))","(2,)",('int32')
3+gather_v2_test_case_1,aclnnGatherV2,"('float16', 'int64', 'float16')","('ND',)",{ 'dim': -1},"((1, 1, 1, 1), (1,), (1, 1, 1, 1))","((0, 0.001), (0, 0))","(2,)",('float16')
4+gather_v2_test_case_2,aclnnGatherV2,"('float32', 'int32', 'float32')","('ND',)",{ 'dim': -2},"((1, 1), (17,), (17, 1))","((-1000, -10), (0, 0))","(2,)",('float32')
5+gather_v2_test_case_3,aclnnGatherV2,"('float32', 'int64', 'float32')","('ND',)",{ 'dim': -2},"((18, 1), (16,), (16, 1))","((-1000, -10), (0, 17))","(2,)",('float32')
6+gather_v2_test_case_4,aclnnGatherV2,"('float16', 'int32', 'float16')","('ND',)",{ 'dim': 1},"((5, 8), (5,), (5, 5))","((-1000, -10), (0, 7))","(2,)",('float16')
@@ -13,14 +13,15 @@ import numpy as np
13 13 
14 14 
15__golden__ = {15__golden__ = {
16- "kernel": {16+ "aclnn": {
17- "index": "index_golden"17+ "aclnnIndex": "aclnn_index_golden",
18- }18+ },
19+ "kernel": {"index": "index_golden"},
19}20}
20 21 
21 22 
22def index_golden(x, mask, out, indices, **kwargs):23def index_golden(x, mask, out, indices, **kwargs):
23- '''24+ """
24 Golden function for index.25 Golden function for index.
25 All the parameters (names and order) follow @index_def.cpp without outputs.26 All the parameters (names and order) follow @index_def.cpp without outputs.
26 All the input Tensors are numpy.ndarray.27 All the input Tensors are numpy.ndarray.
@@ -31,18 +32,18 @@ def index_golden(x, mask, out, indices, **kwargs):
31 32 
32 Returns:33 Returns:
33 Output tensor34 Output tensor
34- '''35+ """
35 import torch36 import torch
36 37 
37 bf16_mark = False38 bf16_mark = False
38 if "bfloat16" in str(x.dtype):39 if "bfloat16" in str(x.dtype):
39 bf16_mark = True40 bf16_mark = True
40 x = x.astype(np.float32)41 x = x.astype(np.float32)
41- 42+ 
42- x_torch = torch.from_numpy(x)43+ x_torch = torch.from_numpy(x) # noqa: F841
43 indices_list = [arr.astype(np.int64) for arr in indices]44 indices_list = [arr.astype(np.int64) for arr in indices]
44- indices_torch = torch.from_numpy(np.array(indices_list))45+ indices_torch = torch.from_numpy(np.array(indices_list)) # noqa: F841
45- 46+ 
46 cmd = "x_torch["47 cmd = "x_torch["
47 idx = 048 idx = 0
48 for i in range(mask.size):49 for i in range(mask.size):
@@ -54,8 +55,29 @@ def index_golden(x, mask, out, indices, **kwargs):
54 cmd += ","55 cmd += ","
55 cmd += "]"56 cmd += "]"
56 res = eval(cmd)57 res = eval(cmd)
57- 58+ 
58 if bf16_mark:59 if bf16_mark:
59 res.to(torch.bfloat16)60 res.to(torch.bfloat16)
60 res = res.numpy()61 res = res.numpy()
61 return res62 return res
63+ 
64+ 
65+def aclnn_index_golden(self, indices, out=None, **kwargs):
66+ """
67+ Aclnn golden for aclnnIndex.
68+ Parameters follow @aclnnIndexGetWorkspaceSize without workspaceSize & executor.
69+ All the input Tensors are torch.Tensor.
70+ """
71+ import torch
72+ 
73+ # Convert indices: empty tensors -> None
74+ if isinstance(indices, (list, tuple)):
75+ idx_list = []
76+ for idx in indices:
77+ if idx is None or idx.numel() == 0:
78+ idx_list.append(None)
79+ else:
80+ idx_list.append(idx.to(torch.int64))
81+ indices = tuple(idx_list)
82+ 
83+ return [torch.ops.aten.index(self, indices)]
@@ -0,0 +1,6 @@
1+api_name,testcase_name,tensor_view_shapes,tensor_dtypes,input_data_ranges,output_tensor_indexes,tensor_formats
2+aclnnIndex,big_shape_aclnnIndex_fp16_int64_ND_random_000003,"((2, 1, 2, 3, 1, 1, 15, 3), ((3, 56628, 5, 2), (3, 56628, 5, 2), (3, 56628, 5, 2), (3, 56628, 5, 2), (3, 56628, 5, 2), (3, 56628, 5, 2), (3, 56628, 5, 2), (3, 56628, 5, 2)), (3, 56628, 5, 2))","('fp16', ('int64', 'int64', 'int64', 'int64', 'int64', 'int64', 'int64', 'int64'), 'fp16')","((-0.01, 0.01), ((-2, 1), (-1, 0), (-2, 1), (-3, 2), (-1, 0), (-1, 0), (-15, 14), (-3, 2)))","(2,)","('ND',)"
3+aclnnIndex,big_shape_aclnnIndex_fp16_int32_ND_random_000004,"((7, 9, 7, 7, 4, 8, 7), ((11, 7, 11, 8, 2, 10), (11, 7, 11, 8, 2, 10), (11, 7, 11, 8, 2, 10), (11, 7, 11, 8, 2, 10), (11, 7, 11, 8, 2, 10), (11, 7, 11, 8, 2, 10), (11, 7, 11, 8, 2, 10)), (11, 7, 11, 8, 2, 10))","('fp16', ('int32', 'int32', 'int32', 'int32', 'int32', 'int32', 'int32'), 'fp16')","((0.01, 1), ((-7, 6), (-9, 8), (-7, 6), (-7, 6), (-4, 3), (-8, 7), (-7, 6)))","(2,)","('ND',)"
4+aclnnIndex,big_shape_aclnnIndex_int32_ND_random_000005,"((53, 83, 138), ((6, 5, 3, 3, 6, 7, 8), (6, 5, 3, 3, 6, 7, 8), (6, 5, 3, 3, 6, 7, 8)), (6, 5, 3, 3, 6, 7, 8))","('int32', ('int32', 'int32', 'int32'), 'int32')","((10, 1000), ((-53, 52), (-83, 82), (-138, 137)))","(2,)","('ND',)"
5+aclnnIndex,big_shape_aclnnIndex_int64_int32_ND_random_000009,"((4, 4, 8, 5, 8, 5), ((3, 6, 6, 4, 6, 6, 5, 3), (3, 6, 6, 4, 6, 6, 5, 3), (3, 6, 6, 4, 6, 6, 5, 3), (3, 6, 6, 4, 6, 6, 5, 3), (3, 6, 6, 4, 6, 6, 5, 3), (3, 6, 6, 4, 6, 6, 5, 3)), (3, 6, 6, 4, 6, 6, 5, 3))","('int64', ('int32', 'int32', 'int32', 'int32', 'int32', 'int32'), 'int64')","((-9223372036854775808, -9223372036854775808), ((-4, 3), (-4, 3), (-8, 7), (-5, 4), (-8, 7), (-5, 4)))","(2,)","('ND',)"
6+aclnnIndex,big_shape_aclnnIndex_int8_int64_ND_random_000013,"((4, 2, 24, 12738, 5), ((247467, 2, 5, 4, 6), (247467, 2, 5, 4, 6), (247467, 2, 5, 4, 6), (247467, 2, 5, 4, 6), (247467, 2, 5, 4, 6)), (247467, 2, 5, 4, 6))","('int8', ('int64', 'int64', 'int64', 'int64', 'int64'), 'int8')","((0, 1), ((-4, 3), (-2, 1), (-24, 23), (-12738, 12737), (-5, 4)))","(2,)","('ND',)"
@@ -11,9 +11,16 @@
11# ----------------------------------------------------------------------------11# ----------------------------------------------------------------------------
12 12 
13import numpy as np13import numpy as np
14+import torch
14from copy import deepcopy15from copy import deepcopy
15 16 
16-__golden__ = {"kernel": {"quant_update_scatter": "quant_update_scatter_golden"}}17+__golden__ = {
18+ "aclnn": {
19+ "aclnnInplaceQuantScatterV2": "aclnn_inplace_quant_scatter_v2_golden",
20+ "aclnnInplaceQuantScatter": "aclnn_inplace_quant_scatter_golden",
21+ },
22+ "kernel": {"quant_update_scatter": "quant_update_scatter_golden"},
23+}
17 24 
18 25 
19def quant_update_scatter_golden(26def quant_update_scatter_golden(
@@ -159,3 +166,169 @@ def quant_update_scatter_golden(
159 output[i][j][k][indices_key + m] = update_value[i][j][k][m]166 output[i][j][k][indices_key + m] = update_value[i][j][k][m]
160 167 
161 return [output]168 return [output]
169+ 
170+ 
171+def aclnn_inplace_quant_scatter_golden(
172+ selfRef,
173+ indices,
174+ updates,
175+ quantScales,
176+ quantZeroPoints,
177+ axis,
178+ quantAxis,
179+ reduction,
180+ **kwargs,
181+):
182+ """
183+ Aclnn golden for aclnnInplaceQuantScatter.
184+ Parameters follow @aclnnInplaceQuantScatterGetWorkspaceSize without workspaceSize & executor.
185+ All the input Tensors are torch.Tensor.
186+ """
187+ import numpy as np
188+ from copy import deepcopy
189+ 
190+ if hasattr(axis, "item"):
191+ axis = axis.item()
192+ if hasattr(quantAxis, "item"):
193+ quantAxis = quantAxis.item()
194+ if hasattr(reduction, "item"):
195+ reduction = reduction.item()
196+ 
197+ def _to_np(t):
198+ if t is None:
199+ return None
200+ if isinstance(t, np.ndarray):
201+ return t
202+ dt = t.dtype
203+ dt_str = str(dt)
204+ if dt == torch.bfloat16:
205+ return t.to(torch.float32).numpy()
206+ if "float8_e5m2" in dt_str:
207+ from ml_dtypes import float8_e5m2 as np_f8_e5m2
208+ 
209+ return t.view(torch.uint8).numpy().view(np_f8_e5m2)
210+ if "float8_e4m3fn" in dt_str:
211+ from ml_dtypes import float8_e4m3fn as np_f8_e4m3
212+ 
213+ return t.view(torch.uint8).numpy().view(np_f8_e4m3)
214+ if "hifloat8" in dt_str:
215+ from en_dtypes import hifloat8 as np_hf8
216+ 
217+ return t.view(torch.uint8).numpy().view(np_hf8)
218+ return t.numpy()
219+ 
220+ var_np = _to_np(selfRef)
221+ indices_np = _to_np(indices)
222+ updates_np = _to_np(updates)
223+ scales_np = _to_np(quantScales)
224+ zp_np = _to_np(quantZeroPoints)
225+ 
226+ result = quant_update_scatter_golden(
227+ deepcopy(var_np),
228+ indices_np,
229+ updates_np,
230+ scales_np,
231+ zp_np,
232+ reduce=reduction,
233+ axis=axis,
234+ quant_axis=quantAxis,
235+ )
236+ 
237+ if isinstance(result, list):
238+ result = result[0]
239+ dt_str = str(selfRef.dtype)
240+ if "float8" in dt_str or "hifloat8" in dt_str:
241+ if isinstance(selfRef, np.ndarray):
242+ return [result]
243+ return [
244+ torch.from_numpy(result.copy().view(np.uint8).copy()).view(selfRef.dtype)
245+ ]
246+ if isinstance(selfRef, np.ndarray):
247+ return [result]
248+ return [torch.from_numpy(np.ascontiguousarray(result.copy())).to(selfRef.dtype)]
249+ 
250+ 
251+def aclnn_inplace_quant_scatter_v2_golden(
252+ selfRef,
253+ indices,
254+ updates,
255+ quantScales,
256+ quantZeroPoints,
257+ axis,
258+ quantAxis,
259+ reduction,
260+ roundMode,
261+ **kwargs,
262+):
263+ """
264+ Aclnn golden for aclnnInplaceQuantScatterV2.
265+ Parameters follow @aclnnInplaceQuantScatterV2GetWorkspaceSize without workspaceSize & executor.
266+ All the input Tensors are torch.Tensor.
267+ """
268+ import numpy as np
269+ from copy import deepcopy
270+ 
271+ if hasattr(axis, "item"):
272+ axis = axis.item()
273+ if hasattr(quantAxis, "item"):
274+ quantAxis = quantAxis.item()
275+ if hasattr(reduction, "item"):
276+ reduction = reduction.item()
277+ if isinstance(roundMode, bytes):
278+ roundMode = roundMode.decode()
279+ elif hasattr(roundMode, "item"):
280+ roundMode = roundMode.item()
281+ 
282+ def _to_np(t):
283+ if t is None:
284+ return None
285+ if isinstance(t, np.ndarray):
286+ return t
287+ dt = t.dtype
288+ dt_str = str(dt)
289+ if dt == torch.bfloat16:
290+ return t.to(torch.float32).numpy()
291+ if "float8_e5m2" in dt_str:
292+ from ml_dtypes import float8_e5m2 as np_f8_e5m2
293+ 
294+ return t.view(torch.uint8).numpy().view(np_f8_e5m2)
295+ if "float8_e4m3fn" in dt_str:
296+ from ml_dtypes import float8_e4m3fn as np_f8_e4m3
297+ 
298+ return t.view(torch.uint8).numpy().view(np_f8_e4m3)
299+ if "hifloat8" in dt_str:
300+ from en_dtypes import hifloat8 as np_hf8
301+ 
302+ return t.view(torch.uint8).numpy().view(np_hf8)
303+ return t.numpy()
304+ 
305+ var_np = _to_np(selfRef)
306+ indices_np = _to_np(indices)
307+ updates_np = _to_np(updates)
308+ scales_np = _to_np(quantScales)
309+ zp_np = _to_np(quantZeroPoints)
310+ 
311+ result = quant_update_scatter_golden(
312+ deepcopy(var_np),
313+ indices_np,
314+ updates_np,
315+ scales_np,
316+ zp_np,
317+ reduce=reduction,
318+ axis=axis,
319+ quant_axis=quantAxis,
320+ round_mode=roundMode if isinstance(roundMode, str) else "rint",
321+ )
322+ 
323+ if isinstance(result, list):
324+ result = result[0]
325+ dt_str = str(selfRef.dtype)
326+ if "float8" in dt_str or "hifloat8" in dt_str:
327+ if isinstance(selfRef, np.ndarray):
328+ return [result]
329+ return [
330+ torch.from_numpy(result.copy().view(np.uint8).copy()).view(selfRef.dtype)
331+ ]
332+ if isinstance(selfRef, np.ndarray):
333+ return [result]
334+ return [torch.from_numpy(np.ascontiguousarray(result.copy())).to(selfRef.dtype)]
@@ -12,7 +12,13 @@
12 12 
13import numpy as np13import numpy as np
14 14 
15-__input__ = {"kernel": {"quant_update_scatter": "quant_update_scatter_input"}}15+__input__ = {
16+ "kernel": {"quant_update_scatter": "quant_update_scatter_input"},
17+ "aclnn": {
18+ "aclnnInplaceQuantScatter": "aclnn_quant_scatter_input",
19+ "aclnnInplaceQuantScatterV2": "aclnn_quant_scatter_v2_input",
20+ },
21+}
16 22 
17 23 
18def quant_update_scatter_input(24def quant_update_scatter_input(
@@ -81,3 +87,96 @@ def quant_update_scatter_input(
81 indices = np.reshape(indices, (shape[0], 2))87 indices = np.reshape(indices, (shape[0], 2))
82 indices = indices.astype(dtype, copy=False)88 indices = indices.astype(dtype, copy=False)
83 return [var, indices, updates, quant_scales, quant_zero_points]89 return [var, indices, updates, quant_scales, quant_zero_points]
90+ 
91+ 
92+def aclnn_quant_scatter_input(*args, **kwargs):
93+ """
94+ Input function for aclnnInplaceQuantScatter.
95+ TTK passes all tensor args; process indices.
96+ """
97+ import torch
98+ import numpy as np
99+ 
100+ var = args[0] if len(args) > 0 else kwargs.get("selfRef")
101+ indices = args[1] if len(args) > 1 else kwargs.get("indices")
102+ updates = args[2] if len(args) > 2 else kwargs.get("updates")
103+ quant_scales = args[3] if len(args) > 3 else kwargs.get("quantScales")
104+ quant_zero_points = args[4] if len(args) > 4 else kwargs.get("quantZeroPoints")
105+ 
106+ axis = kwargs.get("attributes", {}).get("axis", -2)
107+ 
108+ if indices is not None:
109+ import torch
110+ import numpy as np
111+ 
112+ def _to_np(t):
113+ if t is None:
114+ return None
115+ if isinstance(t, np.ndarray):
116+ return t
117+ if not hasattr(t, "dtype"):
118+ return np.array(t)
119+ dt = t.dtype
120+ dt_str = str(dt)
121+ if dt == torch.bfloat16:
122+ return t.to(torch.float32).numpy()
123+ if "float8" in dt_str or "hifloat" in dt_str or "HiFloat" in dt_str:
124+ try:
125+ return t.view(torch.uint8).numpy()
126+ except (TypeError, RuntimeError):
127+ try:
128+ return t.to(torch.uint8).numpy()
129+ except Exception:
130+ return np.frombuffer(t.numpy(), dtype=np.uint8).reshape(t.shape)
131+ try:
132+ return t.numpy()
133+ except (TypeError, RuntimeError):
134+ return t.to(torch.uint8).numpy()
135+ 
136+ var_np = _to_np(var)
137+ upd_np = _to_np(updates)
138+ idx_np = _to_np(indices)
139+ shape, dtype = idx_np.shape, idx_np.dtype
140+ if len(idx_np.shape) == 1:
141+ idx_np = np.random.uniform(
142+ 0, var_np.shape[axis] - upd_np.shape[axis], shape
143+ ).astype(dtype)
144+ else:
145+ idx_list = []
146+ batch_map = {}
147+ while True:
148+ batch = np.random.uniform(0, var_np.shape[0], (1,)).astype(dtype).item()
149+ if batch in batch_map:
150+ avail = set(range(1, var_np.shape[axis] - upd_np.shape[axis]))
151+ for s in batch_map[batch]:
152+ avail -= set(
153+ range(s - upd_np.shape[axis], s + upd_np.shape[axis])
154+ )
155+ if len(avail) > 0:
156+ idx_r = np.random.choice(np.array(list(avail), dtype=object))
157+ batch_map[batch].append(idx_r)
158+ else:
159+ continue
160+ else:
161+ idx_r = np.random.uniform(
162+ 0, var_np.shape[axis] - upd_np.shape[axis], (1,)
163+ ).astype(dtype)
164+ batch_map[batch] = [idx_r.item()]
165+ idx_list.append(batch)
166+ if not isinstance(idx_r, int):
167+ idx_r = idx_r.tolist()[0]
168+ idx_list.append(idx_r)
169+ if len(idx_list) == shape[0] * 2:
170+ break
171+ idx_np = np.reshape(idx_list, (shape[0], 2)).astype(dtype)
172+ indices = torch.from_numpy(idx_np.astype(dtype))
173+ 
174+ return [var, indices, updates, quant_scales, quant_zero_points]
175+ 
176+ 
177+def aclnn_quant_scatter_v2_input(*args, **kwargs):
178+ """
179+ Input function for aclnnInplaceQuantScatterV2.
180+ Same logic as V1.
181+ """
182+ return aclnn_quant_scatter_input(*args, **kwargs)
@@ -0,0 +1,5 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,attributes,absolute_precision
2+InplaceQuantScatter_float32_ND_fuzz_1,aclnnInplaceQuantScatter,"('int8', 'int32', 'float16', 'float32', 'int32')","('ND', 'ND', 'ND', 'ND', 'ND')","((2, 2, 10, 10), (2,), (2, 2, 8, 10), (10,), (10,))","((-10, 10), (1, 1), (-1, 1), (-1, 1), (-1, 1))","{'axis': -2, 'quantAxis' : -1, 'reduction' : 1}",0.0001
3+InplaceQuantScatter_float16_ND_fuzz_2,aclnnInplaceQuantScatter,"('int8', 'int64', 'float16', 'float32', 'int32')","('ND', 'ND', 'ND', 'ND', 'ND')","((2, 2, 10, 10), (2,), (2, 2, 8, 10), (10,), (10,))","((-10, 10), (1, 1), (-1, 1), (-1, 1), (-1, 1))","{'axis': -2, 'quantAxis' : -1, 'reduction' : 1}",0.0001
4+InplaceQuantScatter_float32_ND_fuzz_3,aclnnInplaceQuantScatter,"('int8', 'int32', 'bfloat16', 'bfloat16', 'bfloat16')","('ND', 'ND', 'ND', 'ND', 'ND')","((2, 2, 10, 10), (2,), (2, 2, 8, 10), (10,), (10,))","((-10, 10), (1, 1), (-1, 1), (-1, 1), (-1, 1))","{'axis': -2, 'quantAxis' : -1, 'reduction' : 1}",0.0001
5+InplaceQuantScatter_float16_ND_fuzz_4,aclnnInplaceQuantScatter,"('int8', 'int64', 'bfloat16', 'bfloat16', 'bfloat16')","('ND', 'ND', 'ND', 'ND', 'ND')","((2, 2, 10, 10), (2,), (2, 2, 8, 10), (10,), (10,))","((-10, 10), (1, 1), (-1, 1), (-1, 1), (-1, 1))","{'axis': -2, 'quantAxis' : -1, 'reduction' : 1}",0.0001
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,attributes,absolute_precision,output_inplace_indexes
2+InplaceQuantScatterV2_fuzz_1,aclnnInplaceQuantScatterV2,"('int8', 'int32', 'float16', 'float32', 'int32')","('ND', 'ND', 'ND', 'ND', 'ND')","((2, 2, 10, 10), (2,), (2, 2, 8, 10), (10,), (10,))","((-10, 10), (1, 1), (-1, 1), (-1, 1), (-1, 1))","{'axis': -2, 'quantAxis' : -1, 'reduction' : 1, 'roundMode':'rint'}",0.0001,"(0,)"
3+InplaceQuantScatterV2_fuzz_2,aclnnInplaceQuantScatterV2,"('int8', 'int64', 'float16', 'float32', 'int32')","('ND', 'ND', 'ND', 'ND', 'ND')","((2, 2, 10, 10), (2,), (2, 2, 8, 10), (10,), (10,))","((-10, 10), (1, 1), (-1, 1), (-1, 1), (-1, 1))","{'axis': -2, 'quantAxis' : -1, 'reduction' : 1, 'roundMode':'rint'}",0.0001,"(0,)"
4+InplaceQuantScatterV2_fuzz_3,aclnnInplaceQuantScatterV2,"('int8', 'int32', 'bfloat16', 'bfloat16', 'bfloat16')","('ND', 'ND', 'ND', 'ND', 'ND')","((2, 2, 10, 10), (2,), (2, 2, 8, 10), (10,), (10,))","((-10, 10), (1, 1), (-1, 1), (-1, 1), (-1, 1))","{'axis': -2, 'quantAxis' : -1, 'reduction' : 1, 'roundMode':'rint'}",0.0001,"(0,)"
5+InplaceQuantScatterV2_fuzz_4,aclnnInplaceQuantScatterV2,"('int8', 'int64', 'bfloat16', 'bfloat16', 'bfloat16')","('ND', 'ND', 'ND', 'ND', 'ND')","((2, 2, 10, 10), (2,), (2, 2, 8, 10), (10,), (10,))","((-10, 10), (1, 1), (-1, 1), (-1, 1), (-1, 1))","{'axis': -2, 'quantAxis' : -1, 'reduction' : 1, 'roundMode':'rint'}",0.0001,"(0,)"
6+InplaceQuantScatterV2_fuzz_5,aclnnInplaceQuantScatterV2,"('float8_e4m3fn', 'int32', 'float16', 'float32', 'int32')","('ND', 'ND', 'ND', 'ND', 'ND')","((2, 2, 10, 10), (2,), (2, 2, 8, 10), (10,), (10,))","((-10, 10), (1, 1), (-1, 1), (-1, 1), (-1, 1))","{'axis': -2, 'quantAxis' : -1, 'reduction' : 1, 'roundMode':'rint'}",0.0001,"(0,)"
@@ -11,8 +11,18 @@
11# ----------------------------------------------------------------------------11# ----------------------------------------------------------------------------
12 12 
13import numpy as np13import numpy as np
14+import torch
14 15 
15-__golden__ = {"kernel": {"repeat_interleave": "repeat_interleave_golden"}}16+__golden__ = {
17+ "aclnn": {
18+ "aclnnRepeatInterleaveWithDim": "aclnn_repeat_interleave_with_dim_golden",
19+ "aclnnRepeatInterleaveTensor": "aclnn_repeat_interleave_tensor_golden",
20+ "aclnnRepeatInterleaveIntWithDim": "aclnn_repeat_interleave_int_with_dim_golden",
21+ "aclnnRepeatInterleaveInt": "aclnn_repeat_interleave_int_golden",
22+ "aclnnRepeatInterleave": "aclnn_repeat_interleave_golden",
23+ },
24+ "kernel": {"repeat_interleave": "repeat_interleave_golden"},
25+}
16 26 
17 27 
18def repeat_interleave_golden(x, repeats, *, axis=1000, **kwargs):28def repeat_interleave_golden(x, repeats, *, axis=1000, **kwargs):
@@ -56,3 +66,69 @@ def repeat_interleave_golden(x, repeats, *, axis=1000, **kwargs):
56 if input_dtype.name == "bfloat16":66 if input_dtype.name == "bfloat16":
57 return res_torch.view(torch.int16).numpy().view(x.dtype)67 return res_torch.view(torch.int16).numpy().view(x.dtype)
58 return res_torch.numpy().view(x.dtype)68 return res_torch.numpy().view(x.dtype)
69+ 
70+ 
71+def aclnn_repeat_interleave_golden(self, repeats, outputSize=0, out=None, **kwargs):
72+ """
73+ Aclnn golden for aclnnRepeatInterleave.
74+ Parameters follow @aclnnRepeatInterleaveGetWorkspaceSize without workspaceSize & executor.
75+ All the input Tensors are torch.Tensor.
76+ """
77+ input = self
78+ repeats = repeats
79+ return torch.repeat_interleave(input, repeats)
80+ 
81+ 
82+def aclnn_repeat_interleave_int_golden(
83+ self, repeats=0, outputSize=0, out=None, **kwargs
84+):
85+ """
86+ Aclnn golden for aclnnRepeatInterleaveInt.
87+ Parameters follow @aclnnRepeatInterleaveIntGetWorkspaceSize without workspaceSize & executor.
88+ All the input Tensors are torch.Tensor.
89+ """
90+ input = self
91+ if hasattr(repeats, "item"):
92+ repeats = repeats.item()
93+ return torch.repeat_interleave(input, repeats)
94+ 
95+ 
96+def aclnn_repeat_interleave_int_with_dim_golden(
97+ self, repeats=0, dim=0, outputSize=0, out=None, **kwargs
98+):
99+ """
100+ Aclnn golden for aclnnRepeatInterleaveIntWithDim.
101+ Parameters follow @aclnnRepeatInterleaveIntWithDimGetWorkspaceSize without workspaceSize & executor.
102+ All the input Tensors are torch.Tensor.
103+ """
104+ input = self
105+ if hasattr(repeats, "item"):
106+ repeats = repeats.item()
107+ if hasattr(dim, "item"):
108+ dim = dim.item()
109+ return torch.repeat_interleave(input, repeats, dim)
110+ 
111+ 
112+def aclnn_repeat_interleave_tensor_golden(repeats, outputSize=0, out=None, **kwargs):
113+ """
114+ Aclnn golden for aclnnRepeatInterleaveTensor.
115+ Parameters follow @aclnnRepeatInterleaveTensorGetWorkspaceSize without workspaceSize & executor.
116+ All the input Tensors are torch.Tensor.
117+ """
118+ repeats = repeats
119+ return torch.repeat_interleave(repeats)
atomgit-botatomgit-bot
atomgit-botatomgit-bot9月1日

🟡 Medium Priority

变更行:index/repeat_interleave/tests/assets/golden.py 第113行新增的 aclnn_repeat_interleave_tensor_golden 中调用 torch.repeat_interleave(repeats)。 受影响行为/契约:PyTorch 的 torch.repeat_interleave(input, repeats, dim=None) 中 input 与 repeats 都是必填位置参数(torch/functional.py 中二者均无默认值)。只传一个位置参数会把 repeats 当成 input,随后因缺少 repeats 参数抛出 TypeError: repeat_interleave() missing 1 required positional argument: 'repeats'。即使未来某个版本允许单参数调用,其语义也是按“重复下标”而非“按元素自身计数值重复”,同样与预期输出不符。 失败模式:新增的 CSV 用例 repeat_interleave_test_000012/000013/000014(aclnnRepeatInterleaveTensor,输入 (3,)/(4,)/(5,) 的元素值即重复次数,输出 (6,)/(12,)/(5,))在计算 golden 时必然抛 TypeError,三个测试全部失败。 修复方向:改为 torch.repeat_interleave(repeats, repeats)(可再传入 outputSize 加速),使每个元素按其自身值重复,输出形状与新增用例一致。

建议:将第113行改为 return torch.repeat_interleave(repeats, repeats),即把输入张量同时作为待重复数据和每元素的重复次数,使输入 [2,2,2] 得到 6 个元素的输出。

改动建议
119
- return torch.repeat_interleave(repeats)
119
+ return torch.repeat_interleave(repeats, repeats)
应用建议
likedislike
不准确?
atomgit-botatomgit-bot9月1日

🟠 High Priority

变更行:新增的 aclnn_repeat_interleave_tensor_golden 第 113 行 return torch.repeat_interleave(repeats)。

影响:torch.repeat_interleave 的签名是 repeat_interleave(input, repeats, dim=None, *, output_size=None),repeats 是必填位置参数,没有默认值。当前代码只传了 1 个位置参数(该参数被绑定到 input),调用必然抛出 TypeError: repeat_interleave() missing 1 required positional argument: 'repeats'。且本函数签名只有 repeats 一个张量形参,缺少真正的数据张量。

失败模式:新增 CSV ttk_aclnn_repeat_interleave_with_dim_st.csv 中 000012/000013/000014 三个启用的 aclnnRepeatInterleaveTensor 用例(is_enabled=1)调用该 golden 时都会立即抛 TypeError,测试无法产出期望结果。

建议:补全 repeat_interleave 调用:签名中接收输入张量与 repeats 张量,改为 torch.repeat_interleave(input, repeats)(如需指定输出长度可传 output_size),不能只传一个参数。

likedislike
不准确?
120+ 
121+ 
122+def aclnn_repeat_interleave_with_dim_golden(
123+ self, repeats, dim=0, outputSize=0, out=None, **kwargs
124+):
125+ """
126+ Aclnn golden for aclnnRepeatInterleaveWithDim.
127+ Parameters follow @aclnnRepeatInterleaveWithDimGetWorkspaceSize without workspaceSize & executor.
128+ All the input Tensors are torch.Tensor.
129+ """
130+ input = self
131+ repeats = repeats
132+ if hasattr(dim, "item"):
133+ dim = dim.item()
134+ return torch.repeat_interleave(input, repeats, dim)
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,attributes,tensor_view_shapes,output_tensor_indexes,input_data_ranges,precision_tolerances,strict_precision_mode,is_enabled
2+repeat_interleave_test_000000,aclnnRepeatInterleaveWithDim,"('float32', 'int32', 'float32')","('ND',)","{'dim':0, 'outputSize':6}","((3, 4), (3,), (6, 4))","(2,)","((-100, 100), (2, 2))","((0.001, 0.001),)",,1
3+repeat_interleave_test_000001,aclnnRepeatInterleaveWithDim,"('float32', 'int64', 'float32')","('ND',)","{'dim':0, 'outputSize':9}","((3, 4, 8), (3,), (9, 4, 8))","(2,)","((-100, 100), (3, 3))","((0.001, 0.001),)",,1
4+repeat_interleave_test_000002,aclnnRepeatInterleaveWithDim,"('int32', 'int64', 'int32')","('ND',)","{'dim':1, 'outputSize':20}","((3, 4, 8), (4,), (3, 20, 8))","(2,)","((-100, 100), (5, 5))","((0.001, 0.001),)",,1
5+repeat_interleave_test_000003,aclnnRepeatInterleave,"('float32', 'int32', 'float32')","('ND',)",{'outputSize':24},"((3, 4), (1,), (24,))","(2,)","((-100, 100), (2, 2))","((0.001, 0.001),)",,1
6+repeat_interleave_test_000004,aclnnRepeatInterleave,"('float32', 'int64', 'float32')","('ND',)",{'outputSize':288},"((3, 4, 8), (96,), (288,))","(2,)","((-100, 100), (3, 3))","((0.001, 0.001),)",,1
@@ -11,7 +11,12 @@
11# ----------------------------------------------------------------------------11# ----------------------------------------------------------------------------
12 12 
13 13 
14-__golden__ = {"kernel": {"reverse_v2": "reverse_v2_golden"}}14+__golden__ = {
15+ "aclnn": {
16+ "aclnnFlip": "aclnn_flip_golden",
17+ },
18+ "kernel": {"reverse_v2": "reverse_v2_golden"},
19+}
15 20 
16 21 
17def reverse_v2_golden(x, axis, **kwargs):22def reverse_v2_golden(x, axis, **kwargs):
@@ -36,3 +41,20 @@ def reverse_v2_golden(x, axis, **kwargs):
36 with tf.Session() as sess:41 with tf.Session() as sess:
37 res = sess.run(out, feed_dict={x_holder: x})42 res = sess.run(out, feed_dict={x_holder: x})
38 return res43 return res
44+ 
45+ 
46+def aclnn_flip_golden(self, dims=0, out=None, **kwargs):
47+ """
48+ Aclnn golden for aclnnFlip.
49+ Parameters follow @aclnnFlipGetWorkspaceSize without workspaceSize & executor.
50+ All the input Tensors are torch.Tensor.
51+ """
52+ import torch
53+ 
54+ if hasattr(dims, "item"):
55+ dims = [dims.item()]
56+ elif isinstance(dims, (list, tuple)):
57+ dims = [int(d) for d in dims]
58+ else:
59+ dims = [int(dims)]
60+ return [torch.flip(self, dims)]
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_view_shapes,tensor_formats,tensor_dtypes,attributes,output_tensor_indexes,precision_tolerances,absolute_precision,is_enabled
2+aclnn_flip_test_0000,aclnnFlip,"((512, 2048), (512, 2048))","('ND', 'ND')","('float16', 'float16')",{'dims': [0]},"(-1,)","((0.001, 0.001),)",0,TRUE
3+aclnn_flip_test_0001,aclnnFlip,"((512, 2048), (512, 2048))","('ND', 'ND')","('bfloat16', 'bfloat16')",{'dims': [1]},"(-1,)","((0.001, 0.001),)",0,TRUE
4+aclnn_flip_test_0002,aclnnFlip,"((512, 128), (512, 128))","('ND', 'ND')","('float32', 'float32')",{'dims': [0]},"(-1,)","((0.0001, 0.0001),)",0,TRUE
5+aclnn_flip_test_0003,aclnnFlip,"((512, 128), (512, 128))","('ND', 'ND')","('int8', 'int8')",{'dims': [1]},"(-1,)","((0.0001, 0.0001),)",0,TRUE
6+aclnn_flip_test_0004,aclnnFlip,"((16, 2048), (16, 2048))","('ND', 'ND')","('int32', 'int32')",{'dims': [1]},"(-1,)","((0.0001, 0.0001),)",0,TRUE
@@ -1,3 +1,4 @@
1+import torch
1#!/usr/bin/env python32#!/usr/bin/env python3
2# -*- coding: UTF-8 -*-3# -*- coding: UTF-8 -*-
3# ----------------------------------------------------------------------------4# ----------------------------------------------------------------------------
@@ -11,7 +12,12 @@
11# ----------------------------------------------------------------------------12# ----------------------------------------------------------------------------
12 13 
13 14 
14-__golden__ = {"kernel": {"scatter_elements_v2": "scatter_elements_v2_golden"}}15+__golden__ = {
16+ "aclnn": {
17+ "aclnnScatter": "aclnn_scatter_golden",
18+ },
19+ "kernel": {"scatter_elements_v2": "scatter_elements_v2_golden"},
20+}
15 21 
16 22 
17def scatter_elements_v2_golden(23def scatter_elements_v2_golden(
@@ -53,3 +59,31 @@ def scatter_elements_v2_golden(
53 if "bfloat16" in str(x_dtype):59 if "bfloat16" in str(x_dtype):
54 res = res.astype(x_dtype, copy=False)60 res = res.astype(x_dtype, copy=False)
55 return res61 return res
62+ 
63+ 
64+def aclnn_scatter_golden(self, dim, index, src, reduce, out=None, **kwargs):
65+ """
66+ Aclnn golden for aclnnScatter.
67+ Parameters follow @aclnnScatterGetWorkspaceSize without workspaceSize & executor.
68+ All the input Tensors are torch.Tensor.
69+ """
70+ tensor_x = self
71+ tensor_src = src
72+ 
73+ if hasattr(dim, "item"):
74+ dim = dim.item()
75+ if hasattr(reduce, "item"):
76+ reduce = reduce.item()
77+ 
78+ tensor_index = index.to(torch.int64)
79+ if reduce == 1:
80+ out = torch.scatter_reduce(
81+ tensor_x, dim, tensor_index, tensor_src, reduce="sum"
82+ )
83+ elif reduce == 2:
84+ out = torch.scatter_reduce(
85+ tensor_x, dim, tensor_index, tensor_src, reduce="prod"
86+ )
87+ else:
88+ out = torch.scatter(tensor_x, dim, tensor_index, tensor_src)
89+ return out
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_view_shapes,tensor_dtypes,attributes,output_tensor_indexes,input_data_ranges,precision_tolerances,absolute_precision
2+aclnnScatter_001,aclnnScatter,"((1,), (1,), (1,), (1,))","('float32', 'int32', 'float32', 'float32')","{'dim': -1, 'reduce': 0}","(3,)","((-1, 100), (0, 0), (1, 100))",,
3+aclnnScatter_002,aclnnScatter,"((1,), (1,), (1,), (1,))","('int8', 'int64', 'int8', 'int8')","{'dim': -1, 'reduce': 0}","(3,)","((-1, 100), (0, 0), (1, 100))",,
4+aclnnScatter_003,aclnnScatter,"((1,), (1,), (1,), (1,))","('float16', 'int64', 'float16', 'float16')","{'dim': 0, 'reduce': 0}","(3,)","((-1, 100), (0, 0), (1, 100))",,
5+aclnnScatter_004,aclnnScatter,"((1,), (1,), (1,), (1,))","('double', 'int64', 'double', 'double')","{'dim': 0, 'reduce': 0}","(3,)","((-1, 100), (0, 0), (1, 100))",,
6+aclnnScatter_005,aclnnScatter,"((1,), (1,), (1,), (1,))","('uint8', 'int32', 'uint8', 'uint8')","{'dim': 0, 'reduce': 0}","(3,)","((-1, 100), (0, 0), (1, 100))",,
@@ -11,7 +11,12 @@
11# ----------------------------------------------------------------------------11# ----------------------------------------------------------------------------
12 12 
13 13 
14-__golden__ = {"kernel": {"scatter_update": "scatter_update_golden"}}14+__golden__ = {
15+ "aclnn": {
16+ "aclnnInplaceIndexCopy": "aclnn_inplace_index_copy_golden",
17+ },
18+ "kernel": {"scatter_update": "scatter_update_golden"},
19+}
15 20 
16 21 
17def scatter_update_golden(var, indices, updates, *, use_locking=False, **kwargs):22def scatter_update_golden(var, indices, updates, *, use_locking=False, **kwargs):
@@ -45,3 +50,24 @@ def scatter_update_golden(var, indices, updates, *, use_locking=False, **kwargs)
45 )50 )
46 51 
47 return res52 return res
53+ 
54+ 
55+def aclnn_inplace_index_copy_golden(selfRef, dim, index, source, **kwargs):
56+ """
57+ Aclnn golden for aclnnInplaceIndexCopy.
58+ """
59+ import torch
60+ 
61+ if hasattr(dim, "item"):
62+ dim = dim.item()
63+ 
64+ if selfRef.dtype == torch.bfloat16:
65+ var = selfRef.to(torch.float32)
66+ updates = source.to(torch.float32)
67+ else:
68+ var = selfRef
69+ updates = source
70+ 
71+ indices = index.to(torch.int64)
72+ result = torch.index_copy(var, dim=dim, index=indices, source=updates)
73+ return [result.to(selfRef.dtype)]
@@ -0,0 +1,2 @@
1+testcase_name,api_name,tensor_view_shapes,tensor_formats,tensor_dtypes,output_tensor_indexes,output_inplace_indexes,input_data_ranges,attributes,precision_tolerances,absolute_precision,is_enabled,tensor_storage_shapes,tensor_view_strides,tensor_view_offsets
2+aclnnIndexCopy_float16_int32_ND_lower_boundary_000000,aclnnInplaceIndexCopy,"((1,), (1,), (1,))","('ND', 'ND', 'ND')","['float16', 'int32', 'float16']","(0,)","(0,)","[[-65504.0, 65504.0], [0, 0], [-10, -2]]",{'dim':0},"(0.001, 0.001)",0,True,"((8,), (1,), (1,))","((8,), (1,), (1,))","(0, 0, 0)"
@@ -13,7 +13,12 @@
13import numpy as np13import numpy as np
14 14 
15 15 
16-__golden__ = {"kernel": {"adaptive_avg_pool3d": "adaptive_avg_pool3d_golden"}}16+__golden__ = {
17+ "aclnn": {
18+ "aclnnAdaptiveAvgPool3d": "aclnn_adaptive_avg_pool3d_golden",
19+ },
20+ "kernel": {"adaptive_avg_pool3d": "adaptive_avg_pool3d_golden"},
21+}
17 22 
18 23 
19def adaptive_avg_pool3d_golden(x, output_size, data_format="NDHWC", **kwargs):24def adaptive_avg_pool3d_golden(x, output_size, data_format="NDHWC", **kwargs):
@@ -54,3 +59,22 @@ def adaptive_avg_pool3d_golden(x, output_size, data_format="NDHWC", **kwargs):
54 output = output.numpy().astype(xDtype)59 output = output.numpy().astype(xDtype)
55 60 
56 return output61 return output
62+ 
63+ 
64+def aclnn_adaptive_avg_pool3d_golden(self, outputSize=0, out=None, **kwargs):
65+ """
66+ Aclnn golden for aclnnAdaptiveAvgPool3d.
67+ Parameters follow @aclnnAdaptiveAvgPool3dGetWorkspaceSize without workspaceSize & executor.
68+ All the input Tensors are torch.Tensor.
69+ """
70+ import torch
71+ 
72+ if hasattr(outputSize, "tolist"):
73+ outputSize = outputSize.tolist()
74+ elif isinstance(outputSize, int):
75+ outputSize = [outputSize] * 3
76+ orig_dtype = self.dtype
77+ if orig_dtype in (torch.float16, torch.bfloat16):
78+ self = self.to(torch.float32)
79+ result = torch.nn.functional.adaptive_avg_pool3d(self, outputSize)
80+ return [result.to(orig_dtype)]
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_view_shapes,tensor_formats,tensor_dtypes,output_tensor_indexes,attributes,input_data_ranges
2+aclnnAdaptiveAvgPool3d_bf16_ND_acpw_000001,aclnnAdaptiveAvgPool3d,"((7, 8, 9, 10),(7, 3, 4, 5))","('ND','ND')","('bf16','bf16')","(1,)","{""outputSize"": [3, 4, 5]}","((-1.0, 1.0),(-1.0, 1.0))"
3+aclnnAdaptiveAvgPool3d_fp32_ND_acpw_000002,aclnnAdaptiveAvgPool3d,"((7, 8, 9, 10),(7, 3, 4, 5))","('ND','ND')","('fp32','fp32')","(1,)","{""outputSize"": [3, 4, 5]}","((-1.0, 1.0),(-1.0, 1.0))"
4+aclnnAdaptiveAvgPool3d_fp16_ND_acpw_000003,aclnnAdaptiveAvgPool3d,"((7, 8, 9, 10),(7, 3, 4, 5))","('ND','ND')","('fp16','fp16')","(1,)","{""outputSize"": [3, 4, 5]}","((-1.0, 1.0),(-1.0, 1.0))"
5+aclnnAdaptiveAvgPool3d_bf16_NCDHW_acpw_000003,aclnnAdaptiveAvgPool3d,"((8, 2, 11, 12, 13),(8, 2, 6, 7, 5))","('NCDHW','NCDHW')","('bf16','bf16')","(1,)","{""outputSize"": [6, 7, 5]}","((-1.0, 1.0),(-1.0, 1.0))"
6+aclnnAdaptiveAvgPool3d_fp32_NCDHW_acpw_000004,aclnnAdaptiveAvgPool3d,"((8, 2, 11, 12, 13),(8, 2, 6, 7, 5))","('NCDHW','NCDHW')","('fp32','fp32')","(1,)","{""outputSize"": [6, 7, 5]}","((-1.0, 1.0),(-1.0, 1.0))"
@@ -11,8 +11,14 @@
11# ----------------------------------------------------------------------------11# ----------------------------------------------------------------------------
12 12 
13import numpy as np13import numpy as np
14+import torch
14 15 
15-__golden__ = {"kernel": {"avg_pool3_d": "avg_pool3_d_golden"}}16+__golden__ = {
17+ "aclnn": {
18+ "aclnnAvgPool3d": "aclnn_avg_pool3d_golden",
19+ },
20+ "kernel": {"avg_pool3_d": "avg_pool3_d_golden"},
21+}
16 22 
17 23 
18def get_out_shape(in_width, pad_left, pad_right, kw, dilation, stride, ceil_mode=False):24def get_out_shape(in_width, pad_left, pad_right, kw, dilation, stride, ceil_mode=False):
@@ -197,3 +203,165 @@ def avg_pool3_d_golden(
197 res = np.transpose(res, (0, 2, 3, 4, 1))203 res = np.transpose(res, (0, 2, 3, 4, 1))
198 res = res.astype(input_dtype, copy=False)204 res = res.astype(input_dtype, copy=False)
199 return res205 return res
206+ 
207+ 
208+def aclnn_avg_pool3d_golden(
209+ self,
210+ kernelSize=0,
211+ stride=0,
212+ padding=0,
213+ ceilMode=0,
214+ countIncludePad=0,
215+ divisorOverride=0,
216+ out=None,
217+ **kwargs,
218+):
219+ """
220+ Aclnn golden for aclnnAvgPool3d.
221+ Parameters follow @aclnnAvgPool3dGetWorkspaceSize without workspaceSize & executor.
222+ All the input Tensors are torch.Tensor.
223+ """
224+ inputx = self
225+ input_dtype = inputx.dtype
226+ 
227+ if "float16" in str(input_dtype):
228+ inputx = inputx.to(torch.float32)
229+ attrs = kwargs.get("attributes", {})
230+ ksize = attrs.get("kernelSize", kernelSize)
231+ strides = attrs.get("stride", stride)
232+ pads = attrs.get("padding", padding if padding else [0, 0, 0, 0, 0, 0])
233+ ceil_mode = attrs.get("ceilMode", ceilMode)
234+ countIncludePad = attrs.get("countIncludePad", countIncludePad)
235+ divisor_override = attrs.get("divisorOverride", divisorOverride)
236+ data_format = kwargs.get("tensor_formats", ["NCDHW"])[0]
237+ if divisor_override == 0:
238+ divisor_override = None
239+ 
240+ if len(strides) == 1:
241+ strides = [strides[0], strides[0], strides[0]]
242+ 
243+ if len(ksize) == 1:
244+ ksize = [ksize[0], ksize[0], ksize[0]]
245+ 
246+ if (
247+ len(pads) == 6
248+ and pads[0] == pads[1]
249+ and pads[2] == pads[3]
250+ and pads[4] == pads[5]
251+ ):
252+ pads = [pads[0], pads[2], pads[4]]
253+ 
254+ if data_format.lower() not in ["ncdhw", "nd"]:
255+ raise Exception("AvgPool3d only support ND and NCDHW")
256+ 
257+ if data_format == "ND":
258+ inputx = inputx.reshape(
259+ 1, inputx.shape[0], inputx.shape[1], inputx.shape[2], inputx.shape[3]
260+ )
261+ if len(ksize) == 5:
262+ ksize = [ksize[2], ksize[3], ksize[4]]
263+ if len(strides) == 5:
264+ strides = [strides[2], strides[3], strides[4]]
265+ din, hin, win = inputx.shape[2:]
266+ out = None
267+ if (
268+ len(pads) == 1
269+ and (
270+ pads[0] < ksize[0] / 2 and pads[0] < ksize[1] / 2 and pads[0] < ksize[2] / 2
271+ )
272+ ) or (
273+ len(pads) == 3
274+ and (
275+ pads[0] < ksize[0] / 2 and pads[1] < ksize[1] / 2 and pads[2] < ksize[2] / 2
276+ )
277+ ):
278+ out = torch.nn.functional.avg_pool3d(
279+ inputx,
280+ ksize,
281+ stride=strides,
282+ padding=pads,
283+ ceil_mode=ceil_mode,
284+ count_include_pad=countIncludePad,
285+ divisor_override=divisor_override,
286+ )
287+ else:
288+ if len(pads) == 1:
289+ pads = [pads[0], pads[0], pads[0], pads[0], pads[0], pads[0]]
290+ elif len(pads) == 3:
291+ pads = [pads[0], pads[0], pads[1], pads[1], pads[2], pads[2]]
292+ out_d, out_h, out_w = 0, 0, 0
293+ 
294+ out_d = get_out_shape(
295+ inputx.shape[2], pads[0], pads[1], ksize[0], 1, strides[0], ceil_mode
296+ )
297+ out_h = get_out_shape(
298+ inputx.shape[3], pads[2], pads[3], ksize[1], 1, strides[1], ceil_mode
299+ )
300+ out_w = get_out_shape(
301+ inputx.shape[4], pads[4], pads[5], ksize[2], 1, strides[2], ceil_mode
302+ )
303+ org_pads = pads[:]
304+ 
305+ in_d = (out_d - 1) * strides[0] + (1 * (ksize[0] - 1) + 1)
306+ in_h = (out_h - 1) * strides[1] + (1 * (ksize[1] - 1) + 1)
307+ in_w = (out_w - 1) * strides[2] + (1 * (ksize[2] - 1) + 1)
308+ pads[1] = max(in_d - inputx.shape[2] - pads[0], 0)
309+ pads[3] = max(in_h - inputx.shape[3] - pads[2], 0)
310+ pads[5] = max(in_w - inputx.shape[4] - pads[4], 0)
311+ 
312+ inputx = np.pad(
313+ inputx,
314+ (
315+ (0, 0),
316+ (0, 0),
317+ (pads[0], pads[1]),
318+ (pads[2], pads[3]),
319+ (pads[4], pads[5]),
320+ ),
321+ mode="constant",
322+ constant_values=0,
323+ )
324+ x_tensor = torch.tensor(inputx)
325+ out = torch.nn.functional.avg_pool3d(
326+ x_tensor,
327+ ksize,
328+ stride=strides,
329+ padding=0,
330+ ceil_mode=False,
331+ count_include_pad=False,
332+ divisor_override=1,
333+ )
334+ if (divisor_override is not None) and divisor_override != 0:
335+ out = out / torch.tensor(divisor_override, dtype=torch.float32)
336+ else:
337+ idx = torch.arange(out_d * out_h * out_w)
338+ d_idx = idx // (out_h * out_w)
339+ h_idx = idx % (out_h * out_w) // out_w
340+ w_idx = idx % out_w
341+ d_start = d_idx * strides[0] - org_pads[0]
342+ h_start = h_idx * strides[1] - org_pads[2]
343+ w_start = w_idx * strides[2] - org_pads[4]
344+ d_end = d_start + ksize[0]
345+ h_end = h_start + ksize[1]
346+ w_end = w_start + ksize[2]
347+ if countIncludePad:
348+ d_end.clamp_(max=din + org_pads[1])
349+ h_end.clamp_(max=hin + org_pads[3])
350+ w_end.clamp_(max=win + org_pads[5])
351+ else:
352+ d_start.clamp_(min=0)
353+ h_start.clamp_(min=0)
354+ w_start.clamp_(min=0)
355+ d_end.clamp_(max=din)
356+ h_end.clamp_(max=hin)
357+ w_end.clamp_(max=win)
358+ window_d = d_end - d_start
359+ window_h = h_end - h_start
360+ window_w = w_end - w_start
361+ pool_size = window_d * window_h * window_w
362+ pool_size = pool_size.to(torch.float).reshape(1, 1, out_d, out_h, out_w)
363+ out = out / pool_size
364+ if data_format == "ND":
365+ out = out.reshape(out.shape[1], out.shape[2], out.shape[3], out.shape[4])
366+ res = out.to(input_dtype)
367+ return res
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,attributes,tensor_view_shapes,scalar_dtypes,scalar_data_ranges,output_tensor_indexes,input_data_ranges,precision_tolerances,is_enabled
2+aclnnAvgPool3d_bf16_ND_acpw_000003,aclnnAvgPool3d,"('bfloat16',)","('ND',)","{'kernelSize': [3, 2, 15], 'stride': [15, 50, 127], 'padding': [0, 1, 6], 'ceilMode': False, 'countIncludePad': False, 'divisorOverride': 844}","((52, 3, 2, 39), (52, 1, 1, 1))",(),"((None, None),)","(1,)","((-1, 1),)",,TRUE
3+aclnnAvgPool3d_bf16_ND_acpw_000010,aclnnAvgPool3d,"('bfloat16',)","('ND',)","{'kernelSize': [2], 'stride': [89], 'padding': [1], 'ceilMode': False, 'countIncludePad': True, 'divisorOverride': 749}","((4, 7, 12, 3), (4, 1, 1, 1))",(),"((None, None),)","(1,)","((-2, -1),)",,TRUE
4+aclnnAvgPool3d_fp32_NCDHW_acpw_000357,aclnnAvgPool3d,"('float32',)","('NCDHW',)","{'kernelSize': [3, 6, 290], 'stride': [73, 72, 173], 'padding': [0, 3, 58], 'ceilMode': True, 'countIncludePad': False, 'divisorOverride': 619}","((3, 3, 3, 6, 638), (3, 3, 1, 1, 4))",(),"((None, None),)","(1,)","((-0.01, 0.01),)",,TRUE
5+aclnnAvgPool3d_fp32_NCDHW_acpw_000358,aclnnAvgPool3d,"('float32',)","('NCDHW',)","{'kernelSize': [4, 5, 4], 'stride': [100, 46, 84], 'padding': [0, 2, 0], 'ceilMode': False, 'countIncludePad': True, 'divisorOverride': 551}","((120, 3, 5, 5, 18), (120, 3, 1, 1, 1))",(),"((None, None),)","(1,)","((-100, 100),)",,TRUE
6+aclnnAvgPool3d_fp32_ND_acpw_000066,aclnnAvgPool3d,"('float32',)","('ND',)","{'kernelSize': [2], 'stride': [65], 'padding': [1], 'ceilMode': False, 'countIncludePad': True, 'divisorOverride': 75}","((3, 2915, 3, 4), (3, 45, 1, 1))",(),"((None, None),)","(1,)","((0.01, 1),)",,TRUE
@@ -14,7 +14,10 @@ import numpy as np
14from copy import deepcopy14from copy import deepcopy
15 15 
16__golden__ = {16__golden__ = {
17- "kernel": {"max_pool3d_grad_with_argmax": "max_pool3d_grad_with_argmax_golden"}17+ "aclnn": {
18+ "aclnnMaxPool2dWithIndicesBackward": "aclnn_max_pool2d_with_indices_backward_golden",
19+ },
20+ "kernel": {"max_pool3d_grad_with_argmax": "max_pool3d_grad_with_argmax_golden"},
18}21}
19 22 
20 23 
@@ -92,3 +95,58 @@ def max_pool3d_grad_with_argmax_golden(
92 backward_out = backward_out.transpose(0, 2, 3, 4, 1)95 backward_out = backward_out.transpose(0, 2, 3, 4, 1)
93 96 
94 return backward_out97 return backward_out
98+ 
99+ 
100+def aclnn_max_pool2d_with_indices_backward_golden(
101+ gradOutput,
102+ self,
103+ indices,
104+ kernelSize=0,
105+ stride=0,
106+ padding=0,
107+ dilation=0,
108+ ceilMode=0,
109+ gradInput=None,
110+ **kwargs,
111+):
112+ """
113+ Aclnn golden for aclnnMaxPool2dWithIndicesBackward.
114+ Parameters follow @aclnnMaxPool2dWithIndicesBackwardGetWorkspaceSize without workspaceSize & executor.
115+ All the input Tensors are torch.Tensor.
116+ """
117+ import torch
118+ 
119+ grad = gradOutput
120+ grad_dtype = grad.dtype
121+ indices_dtype = indices.dtype
122+ 
123+ input_grad_format = kwargs.get("tensor_formats", ["NCHW"])[0]
124+ if grad_dtype == torch.float16 or grad_dtype == torch.bfloat16:
125+ grad = grad.to(torch.float32)
126+ self = self.to(torch.float32)
127+ if indices_dtype == torch.int32:
128+ indices = indices.to(torch.int64)
129+ if input_grad_format == "NHWC":
130+ grad = grad.permute(0, 3, 1, 2)
131+ self = self.permute(0, 3, 1, 2)
132+ indices = indices.permute(0, 3, 1, 2)
133+ 
134+ ceilMode = bool(ceilMode) if ceilMode else False
135+ 
136+ backward_out = torch.ops.aten.max_pool2d_with_indices_backward(
137+ grad,
138+ self,
139+ kernel_size=kernelSize,
140+ stride=stride,
141+ padding=padding,
142+ dilation=dilation,
143+ ceil_mode=ceilMode,
144+ indices=indices,
145+ )
146+ if grad_dtype == torch.float16:
147+ backward_out = backward_out.to(torch.float16)
148+ elif grad_dtype == torch.bfloat16:
149+ backward_out = backward_out.to(torch.bfloat16)
150+ if input_grad_format == "NHWC":
151+ backward_out = backward_out.permute(0, 2, 3, 1)
152+ return backward_out
@@ -0,0 +1,97 @@
1+#!/usr/bin/env python3
2+# -*- coding: UTF-8 -*-
3+# ----------------------------------------------------------------------------
4+# Copyright (c) 2026 Huawei Technologies Co., Ltd.
5+# This program is free software, you can redistribute it and/or modify it under the terms and conditions of
6+# CANN Open Software License Agreement Version 2.0 (the "License").
7+# Please refer to the License for details. You may not use this file except in compliance with the License.
8+# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
9+# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
10+# See LICENSE in the root of the software repository for the full text of the License.
11+# ----------------------------------------------------------------------------
12+ 
13+import numpy as np
14+from functools import reduce
15+import operator
16+ 
17+__input__ = {
18+ "aclnn": {"aclnnMaxPool2dWithIndicesBackward": "aclnn_max_pool_backward_input"},
19+}
20+ 
21+ 
22+def aclnn_max_pool_backward_input(*args, **kwargs):
23+ """
24+ Input function for aclnnMaxPool2dWithIndicesBackward.
25+ Generate valid self, gradOutput and indices via forward maxpool.
26+ Tensors: gradOutput(0), self(1), indices(2), gradInput(3)
27+ 
28+ Following opstest approach: generate random self, run forward maxpool,
29+ use forward output as gradOutput, forward indices as indices.
30+ """
31+ import torch
32+ import torch.nn.functional as F
33+ 
34+ grad_output = args[0]
35+ self_input = args[1]
36+ indices = args[2]
37+ 
38+ attrs = kwargs.get("attributes", {})
39+ kernel_size = attrs.get("kernelSize", [2, 2])
40+ stride = attrs.get("stride", [2, 2])
41+ padding = attrs.get("padding", [1, 1])
42+ dilation = attrs.get("dilation", [1, 1])
43+ ceil_mode = attrs.get("ceilMode", False)
44+ 
45+ input_grad_format = kwargs.get("tensor_formats", ["NCHW"])[0]
46+ 
47+ # Generate random self data (like opstest: random int8 cast to float)
48+ self_shape = self_input.shape
49+ ele_num = reduce(operator.mul, self_shape)
50+ random_array = np.random.randint(
51+ low=np.iinfo(np.int8).min,
52+ high=np.iinfo(np.int8).max + 1,
53+ size=(ele_num,),
54+ dtype=np.int8,
55+ )
56+ input_x = random_array.reshape(self_shape)
57+ 
58+ if self_input.dtype == torch.float16 or self_input.dtype == torch.bfloat16:
59+ input_x = input_x.astype(np.float32)
60+ x_torch = torch.from_numpy(input_x)
61+ else:
62+ input_x = input_x.astype(np.float32)
63+ x_torch = torch.from_numpy(input_x)
64+ 
65+ # NHWC -> NCHW for forward maxpool
66+ if input_grad_format == "NHWC":
67+ x_torch = x_torch.permute(0, 3, 1, 2)
68+ 
69+ max_out, max_indices = F.max_pool2d_with_indices(
70+ x_torch,
71+ kernel_size=kernel_size,
72+ stride=stride,
73+ padding=padding,
74+ dilation=dilation,
75+ ceil_mode=bool(ceil_mode),
76+ )
77+ 
78+ # Convert back to original dtype
79+ orig_dtype = self_input.dtype
80+ if orig_dtype == torch.float16 or orig_dtype == torch.bfloat16:
81+ max_out = max_out.to(orig_dtype)
82+ x_torch = x_torch.to(orig_dtype)
83+ 
84+ # Convert indices to expected dtype
85+ indices_dtype = indices.dtype
86+ max_indices = max_indices.to(indices_dtype)
87+ 
88+ # NHWC -> back
89+ if input_grad_format == "NHWC":
90+ x_torch = x_torch.permute(0, 2, 3, 1)
91+ max_out = max_out.permute(0, 2, 3, 1)
92+ max_indices = max_indices.permute(0, 2, 3, 1)
93+ 
94+ # Copy results into existing tensors in-place
95+ self_input.copy_(x_torch)
96+ grad_output.copy_(max_out)
97+ indices.copy_(max_indices)
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_view_shapes,tensor_formats,tensor_dtypes,output_tensor_indexes,attributes,input_data_ranges,is_enabled
2+MaxPoolGradWithArgmaxV3_fp32_int32_NCHW_infnan_random_0000140,aclnnMaxPool2dWithIndicesBackward,"((1, 64, 5, 16), (1, 64, 9, 30), (1, 64, 5, 16), (1, 64, 9, 30))","(""NCHW"",""NCHW"",""NCHW"",""NCHW"",)","('float32', 'float32', 'int32', 'float32')","(3,)","{""kernelSize"": [2, 2], ""stride"": [2, 2], ""padding"": [1, 1], ""dilation"": [1, 1], ""ceilMode"": False}","((10,1000), (10,1000), (0, 5814))",TRUE
3+MaxPoolGradWithArgmaxV3_fp16_int32_NCHW_infnan_random_0000141,aclnnMaxPool2dWithIndicesBackward,"((1, 64, 3, 83), (1, 64, 5, 165), (1, 64, 3, 83),(1, 64, 5, 165))","(""NCHW"",""NCHW"",""NCHW"",""NCHW"",)","('float16', 'float16', 'int32', 'float16')","(3,)","{""kernelSize"": [2, 2], ""stride"": [2, 2], ""padding"": [1, 1], ""dilation"": [1, 1], ""ceilMode"": False}","((10,1000), (10,1000), (0, 5814))",TRUE
4+MaxPoolGradWithArgmaxV3_bf16_int32_NCHW_infnan_random_0000142,aclnnMaxPool2dWithIndicesBackward,"((2, 64, 3, 3), (2, 64, 5, 5), (2, 64, 3, 3),(2, 64, 5, 5))","(""NCHW"",""NCHW"",""NCHW"",""NCHW"",)","('bfloat16', 'bfloat16', 'int32', 'bfloat16')","(3,)","{'kernelSize': [2, 2], 'stride': [2, 2], 'padding': [1, 1], 'dilation': [1, 1], 'ceilMode': False}","((10,1000), (10,1000), (0, 5814))",TRUE
5+MaxPoolGradWithArgmaxV3_fp32_int64_NCHW_infnan_random_0000140,aclnnMaxPool2dWithIndicesBackward,"((1, 64, 5, 16), (1, 64, 9, 30), (1, 64, 5, 16), (1, 64, 9, 30))","(""NCHW"",""NCHW"",""NCHW"",""NCHW"",)","('float32', 'float32', 'int64', 'float32')","(3,)","{""kernelSize"": [2, 2], ""stride"": [2, 2], ""padding"": [1, 1], ""dilation"": [1, 1], ""ceilMode"": False}","((10,1000), (10,1000), (0, 5814))",TRUE
6+MaxPoolGradWithArgmaxV3_fp16_int64_NCHW_infnan_random_0000141,aclnnMaxPool2dWithIndicesBackward,"((1, 64, 3, 83), (1, 64, 5, 165), (1, 64, 3, 83),(1, 64, 5, 165))","(""NCHW"",""NCHW"",""NCHW"",""NCHW"",)","('float16', 'float16', 'int64', 'float16')","(3,)","{""kernelSize"": [2, 2], ""stride"": [2, 2], ""padding"": [1, 1], ""dilation"": [1, 1], ""ceilMode"": False}","((10,1000), (10,1000), (0, 5814))",TRUE
@@ -14,7 +14,11 @@ import numpy as np
14from copy import deepcopy14from copy import deepcopy
15 15 
16__golden__ = {16__golden__ = {
17- "kernel": {"max_pool3d_with_argmax_v2": "max_pool3d_with_argmax_v2_golden"}17+ "aclnn": {
18+ "aclnnMaxPool3dWithArgmax": "aclnn_max_pool3d_with_argmax_golden",
19+ "aclnnMaxPool2dWithIndices": "aclnn_max_pool2d_with_indices_golden",
20+ },
21+ "kernel": {"max_pool3d_with_argmax_v2": "max_pool3d_with_argmax_v2_golden"},
18}22}
19 23 
20 24 
@@ -120,3 +124,144 @@ def max_pool3d_with_argmax_v2_golden(
120 if out_argmax_format == "NDHWC":124 if out_argmax_format == "NDHWC":
121 out_argmax = out_argmax.transpose(0, 2, 3, 4, 1)125 out_argmax = out_argmax.transpose(0, 2, 3, 4, 1)
122 return out_y, out_argmax126 return out_y, out_argmax
127+ 
128+ 
129+def aclnn_max_pool3d_with_argmax_golden(
130+ self,
131+ kernelSize=0,
132+ stride=0,
133+ padding=0,
134+ dilation=0,
135+ ceilMode=0,
136+ out=None,
137+ indices=None,
138+ **kwargs,
139+):
140+ """
141+ Aclnn golden for aclnnMaxPool3dWithArgmax.
142+ Parameters follow @aclnnMaxPool3dWithArgmaxGetWorkspaceSize without workspaceSize & executor.
143+ All the input Tensors are torch.Tensor.
144+ """
145+ import torch
146+ import torch.nn.functional as F
147+ from copy import deepcopy
148+ 
149+ input_x = deepcopy(self)
150+ inpu_x_dtype = input_x.dtype
151+ input_x = input_x.to(torch.float32)
152+ input_x_format = kwargs.get("tensor_formats", ["NCDHW"])[0]
153+ output_indices = deepcopy(indices) if indices is not None else None
154+ output_indices_dtype = (
155+ output_indices.dtype if output_indices is not None else torch.int64
156+ )
157+ if input_x_format == "NDHWC":
158+ input_x = input_x.permute(0, 4, 1, 2, 3)
159+ elif input_x_format == "NHWC":
160+ input_x = input_x.permute(3, 0, 1, 2)
161+ 
162+ attr = {"return_indices": True}
163+ attr["kernel_size"] = kwargs.get("attributes", {}).get("kernelSize", kernelSize)
164+ attr["stride"] = kwargs.get("attributes", {}).get("stride", stride)
165+ attr["padding"] = kwargs.get("attributes", {}).get("padding", padding)
166+ attr["dilation"] = kwargs.get("attributes", {}).get("dilation", dilation)
167+ attr["ceil_mode"] = kwargs.get("attributes", {}).get("ceilMode", ceilMode)
168+ 
169+ out_y, out_argmax = F.max_pool3d(input_x, **attr)
170+ 
171+ out_y = out_y.to(inpu_x_dtype)
172+ out_argmax = out_argmax.to(output_indices_dtype)
173+ if input_x_format == "NDHWC":
174+ out_y = out_y.permute(0, 2, 3, 4, 1)
175+ out_argmax = out_argmax.permute(0, 2, 3, 4, 1)
176+ elif input_x_format == "NHWC":
177+ out_y = out_y.permute(1, 2, 3, 0)
178+ out_argmax = out_argmax.permute(1, 2, 3, 0)
179+ 
180+ return out_y, out_argmax
181+ 
182+ 
183+def aclnn_max_pool2d_with_indices_golden(
184+ self,
185+ kernelSize=0,
186+ stride=0,
187+ padding=0,
188+ dilation=0,
189+ ceilMode=0,
190+ out=None,
191+ indices=None,
192+ **kwargs,
193+):
194+ """
195+ Aclnn golden for aclnnMaxPool2dWithIndices.
196+ Parameters follow @aclnnMaxPool2dWithIndicesGetWorkspaceSize without workspaceSize & executor.
197+ All the input Tensors are torch.Tensor.
198+ """
199+ import torch
200+ import torch.nn as nn
201+ from copy import deepcopy
202+ 
203+ input_x = deepcopy(self)
204+ 
205+ inpu_x_dtype = input_x.dtype
206+ 
207+ input_x_format = kwargs.get("tensor_formats", ["NCHW"])[0]
208+ output_indices = deepcopy(indices) if indices is not None else None
209+ output_indices_dtype = (
210+ output_indices.dtype if output_indices is not None else torch.int64
211+ )
212+ attrs = kwargs.get("attributes", {})
213+ attr_re_ksize = attrs.get("kernelSize", kernelSize) if attrs else kernelSize
214+ if isinstance(attr_re_ksize, list) and len(attr_re_ksize) == 4:
215+ attr_re_ksize = [attr_re_ksize[1], attr_re_ksize[2]]
216+ attr_re_strides = attrs.get("stride", stride) if attrs else stride
217+ if isinstance(attr_re_strides, list) and len(attr_re_strides) == 4:
218+ attr_re_strides = [attr_re_strides[1], attr_re_strides[2]]
219+ attr_re_pads = attrs.get("padding", padding) if attrs else padding
220+ if isinstance(attr_re_pads, list) and len(attr_re_pads) == 4:
221+ attr_re_pads = [attr_re_pads[1], attr_re_pads[2]]
222+ 
223+ attr_op_dilations = attrs.get("dilation", dilation) if attrs else dilation
224+ if isinstance(attr_op_dilations, list) and len(attr_op_dilations) == 4:
225+ attr_op_dilations = [attr_op_dilations[1], attr_op_dilations[2]]
226+ if attr_op_dilations == 0:
227+ attr_op_dilations = None
228+ 
229+ attr_op_ceil_mode = attrs.get("ceilMode", ceilMode) if attrs else ceilMode
230+ if attr_op_ceil_mode == 0:
231+ attr_op_ceil_mode = False
232+ 
233+ if input_x_format == "NHWC":
234+ input_x = input_x.permute(0, 3, 1, 2)
235+ 
236+ if "float16" == str(inpu_x_dtype) or "bfloat16" == str(inpu_x_dtype):
237+ input_x = input_x.to(torch.float32)
238+ 
239+ attr = {
240+ "kernel_size": attr_re_ksize,
241+ "stride": attr_re_strides,
242+ "padding": attr_re_pads,
243+ }
244+ if attr_op_dilations is not None:
245+ attr["dilation"] = attr_op_dilations
246+ if attr_op_ceil_mode is not None or attr_op_ceil_mode is False:
247+ attr["ceil_mode"] = bool(attr_op_ceil_mode) if attr_op_ceil_mode else False
248+ attr["return_indices"] = True
249+ cpuMaxPool = nn.MaxPool2d(**attr)
250+ max_out, max_indices = cpuMaxPool(input_x)
251+ 
252+ if "bfloat16" == str(inpu_x_dtype):
253+ out_y = max_out.to(inpu_x_dtype, copy=False)
254+ else:
255+ out_y = max_out.to(inpu_x_dtype)
256+ if input_x_format == "NHWC":
257+ out_y = out_y.permute(0, 2, 3, 1)
258+ 
259+ if str(output_indices_dtype) == "torch.int32":
260+ out_argmax = max_indices.to(torch.int32)
261+ else:
262+ out_argmax = max_indices.to(torch.int64)
263+ 
264+ if input_x_format == "NHWC":
265+ out_argmax = out_argmax.permute(0, 2, 3, 1)
266+ 
267+ return out_y, out_argmax
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_view_shapes,tensor_formats,tensor_dtypes,output_tensor_indexes,attributes,input_data_ranges,is_enabled
2+aclnn_nchw_int32,aclnnMaxPool2dWithIndices,"((4, 16, 32, 32), (4, 16, 16, 16), (4, 16, 16, 16))","(""NCHW"",""NCHW"",""NCHW"",)","('float32', 'float32', 'int32',)","(1,2)","{'kernelSize':[2, 2], 'stride':[2, 2], 'padding':[0 ,0], 'dilation':[1,1], 'ceilMode':True}","((10,1000), (10,1000), (0, 255))",TRUE
3+aclnn_nhwc_int32,aclnnMaxPool2dWithIndices,"((4, 10, 10, 16), (4, 5, 5, 16), (4, 5, 5, 16))","(""NHWC"",""NHWC"",""NHWC"",)","('float32', 'float32', 'int32',)","(1,2)","{'kernelSize':[2, 2], 'stride':[2, 2], 'padding':[0 ,0], 'dilation':[1,1], 'ceilMode':True}","((10,1000), (10,1000), (0, 255))",TRUE
4+aclnn_nchw_int64,aclnnMaxPool2dWithIndices,"((4, 16, 32, 32), (4, 16, 16, 16), (4, 16, 16, 16))","(""NCHW"",""NCHW"",""NCHW"",)","('float16', 'float16', 'int64',)","(1,2)","{'kernelSize':[2, 2], 'stride':[2, 2], 'padding':[0 ,0], 'dilation':[1,1], 'ceilMode':True}","((10,1000), (10,1000), (0, 255))",TRUE
5+aclnn_nhwc_int64,aclnnMaxPool2dWithIndices,"((4, 10, 10, 16), (4, 5, 5, 16), (4, 5, 5, 16))","(""NHWC"",""NHWC"",""NHWC"",)","('bfloat16', 'bfloat16', 'int64',)","(1,2)","{'kernelSize':[2, 2], 'stride':[2, 2], 'padding':[0 ,0], 'dilation':[1,1], 'ceilMode':True}","((10,1000), (10,1000), (0, 255))",TRUE
6+aclnn_nchw_dilation,aclnnMaxPool2dWithIndices,"((1, 1, 120, 120), (1, 1, 59, 59), (1, 1, 59, 59))","(""NCHW"",""NCHW"",""NCHW"",)","('float32', 'float32', 'int32',)","(1,2)","{'kernelSize':[2, 2], 'stride':[2, 2], 'padding':[0 ,0], 'dilation':[2,2], 'ceilMode':False}","((10,1000), (10,1000), (0, 255))",TRUE
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_view_shapes,tensor_formats,tensor_dtypes,output_tensor_indexes,attributes,input_data_ranges
2+aclnnMaxPool3dWithArgmax_NCDHW_000001,aclnnMaxPool3dWithArgmax,"((2, 3, 4, 8, 16),(2, 3, 3, 5, 9),(2, 3, 3, 5, 9))","('NCDHW','NCDHW','NCDHW')","('fp32','fp32','int32')","(1,2,)","{""kernelSize"": [2], ""stride"": [], ""padding"": [1], ""dilation"": [1], ""ceilMode"": True}","((-1.0, 1.0),(-1.0, 1.0),(0,1))"
3+aclnnMaxPool3dWithArgmax_NCDHW_000002,aclnnMaxPool3dWithArgmax,"((3, 4, 8, 16),(3, 3, 5, 9),(3, 3, 5, 9))","('NCHW','NCHW','NCHW')","('fp16','fp16','int32')","(1,2,)","{""kernelSize"": [2, 2, 2], ""stride"": [2], ""padding"": [1, 1, 1], ""dilation"": [1, 1, 1], ""ceilMode"": False}","((-1.0, 1.0),(-1.0, 1.0),(0,1))"
4+aclnnMaxPool3dWithArgmax_NCDHW_000003,aclnnMaxPool3dWithArgmax,"((2, 3, 4, 8, 16),(2, 3, 3, 5, 6),(2, 3, 3, 5, 6))","('NCDHW','NCDHW','NCDHW')","('bf16','bf16','int32')","(1,2,)","{""kernelSize"": [2, 4, 8], ""stride"": [2, 2, 2], ""padding"": [1, 2, 1], ""dilation"": [1, 1, 1], ""ceilMode"": False}","((-1.0, 1.0),(-1.0, 1.0),(0,1))"
5+aclnnMaxPool3dWithArgmax_NDHWC_000001,aclnnMaxPool3dWithArgmax,"((2, 3, 4, 8, 16),(2, 2, 3, 5, 16),(2, 2, 3, 5, 16))","('NDHWC','NDHWC','NDHWC')","('fp32','fp32','int64')","(1,2,)","{""kernelSize"": [2], ""stride"": [], ""padding"": [1], ""dilation"": [2], ""ceilMode"": True}","((-1.0, 1.0),(-1.0, 1.0),(0,1))"
6+aclnnMaxPool3dWithArgmax_NDHWC_000002,aclnnMaxPool3dWithArgmax,"((2, 3, 4, 8, 16),(2, 2, 2, 4, 16),(2, 2, 2, 4, 16))","('NDHWC','NDHWC','NDHWC')","('fp16','fp16','int64')","(1,2,)","{""kernelSize"": [2, 2, 2], ""stride"": [2], ""padding"": [1, 1, 1], ""dilation"": [2, 2, 2], ""ceilMode"": False}","((-1.0, 1.0),(-1.0, 1.0),(0,1))"
@@ -12,7 +12,12 @@
12 12 
13import numpy as np13import numpy as np
14 14 
15-__golden__ = {"kernel": {"max_pool_v3": "max_pool_v3_golden"}}15+__golden__ = {
16+ "aclnn": {
17+ "aclnnMaxPool": "aclnn_max_pool_golden",
18+ },
19+ "kernel": {"max_pool_v3": "max_pool_v3_golden"},
20+}
16 21 
17 22 
18def max_pool_v3_golden(23def max_pool_v3_golden(
@@ -128,3 +133,55 @@ def max_pool_v3_golden(
128 if data_format == "NCHW":133 if data_format == "NCHW":
129 result = np.transpose(result, (0, 3, 1, 2))134 result = np.transpose(result, (0, 3, 1, 2))
130 return result.astype(input_dtype, copy=False)135 return result.astype(input_dtype, copy=False)
136+ 
137+ 
138+def aclnn_max_pool_golden(
139+ self,
140+ kernelShape=0,
141+ strides=0,
142+ autoPad=0,
143+ pads=0,
144+ dilations=0,
145+ ceilMode=0,
146+ out=None,
147+ **kwargs,
148+):
149+ """
150+ Aclnn golden for aclnnMaxPool.
151+ Parameters follow @aclnnMaxPoolGetWorkspaceSize without workspaceSize & executor.
152+ All the input Tensors are torch.Tensor.
153+ """
154+ import torch
155+ 
156+ x = self
157+ attrs = kwargs.get("attributes", {})
158+ kernel_size = attrs.get("kernelShape", kernelShape) if attrs else kernelShape
159+ stride = attrs.get("strides", strides) if attrs else strides
160+ padding = attrs.get("pads", pads) if attrs else pads
161+ dilation = attrs.get("dilations", dilations) if attrs else dilations
162+ ceil_mode = attrs.get("ceilMode", ceilMode) if attrs else ceilMode
163+ tensor_format = kwargs.get("tensor_formats", ["NCHW"])[0]
164+ 
165+ is_nhwc = False
166+ if tensor_format.lower() == "nhwc":
167+ x = torch.permute(x, dims=[0, 2, 3, 1])
168+ is_nhwc = True
169+ 
170+ input_dtype = x.dtype
171+ if "float16" in str(input_dtype):
172+ x = x.to(torch.float32)
173+ 
174+ res = torch.nn.functional.max_pool2d(
175+ x,
176+ kernel_size,
177+ stride=stride,
178+ padding=padding,
179+ dilation=dilation,
180+ ceil_mode=bool(ceil_mode) if ceil_mode else False,
181+ return_indices=False,
182+ )
183+ 
184+ res = res.to(input_dtype)
185+ if is_nhwc:
186+ res = torch.permute(res, dims=[0, 3, 1, 2])
187+ return res
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,attributes,tensor_view_shapes,output_tensor_indexes,input_data_ranges
2+aclnnMaxPool_random_000001,aclnnMaxPool,"('float32', 'float32')","('NCHW', 'NCHW')","{'kernelShape': [2,2], 'strides': [2,2], 'autoPad' :0, 'pads' : [0], 'dilations': [1], 'ceilMode':False}","((1,1,2,2), (1,1,1,1))","(1,)","((0.001, 0.01),)"
3+aclnnMaxPool_random_000002,aclnnMaxPool,"('float16', 'float16')","('NCHW', 'NCHW')","{'kernelShape': [2,2], 'strides': [2,2], 'autoPad' :0, 'pads' : [0], 'dilations': [1], 'ceilMode':False}","((1,1,2,2), (1,1,1,1))","(1,)","((-1000, -10),)"
4+aclnnMaxPool_random_000003,aclnnMaxPool,"('bfloat16', 'bfloat16')","('NCHW', 'NCHW')","{'kernelShape': [2,2], 'strides': [2,2], 'autoPad' :0, 'pads' : [0], 'dilations': [1], 'ceilMode':False}","((1,1,2,2), (1,1,1,1))","(1,)","((1, 2),)"
5+aclnnMaxPool_random_000004,aclnnMaxPool,"('float32', 'float32')","('NCHW', 'NCHW')","{'kernelShape': [2,2], 'strides': [2,2], 'autoPad' :0, 'pads' : [0], 'dilations': [1], 'ceilMode':False}","((64,1,16,16), (64,1,8,8))","(1,)","((-1, -0.01),)"
6+aclnnMaxPool_random_000005,aclnnMaxPool,"('float32', 'float32')","('NCHW', 'NCHW')","{'kernelShape': [3], 'strides': [3], 'autoPad' :0, 'pads' : [1], 'dilations': [1], 'ceilMode':True}","((16,1,9,9), (16,1,4,4))","(1,)","((-0.01, 0.01),)"
@@ -256,25 +256,51 @@ def dynamic_quant_golden(
256 )256 )
257 257 
258 258 
259-def aclnn_dynamic_quant_golden(x, smoothScalesOptional, yOut, scaleOut, **kwargs):259+def aclnn_dynamic_quant_golden(
260+ x, smoothScalesOptional, yOut=None, scaleOut=None, **kwargs
261+):
260 """262 """
261 Aclnn golden for aclnnDynamicQuant.263 Aclnn golden for aclnnDynamicQuant.
264+ Parameters follow @aclnnDynamicQuantGetWorkspaceSize without workspaceSize & executor.
265+ All the input Tensors are torch.Tensor.
262 """266 """
263- x_f = x.to(torch.float32) if x.dtype != torch.float32 else x267+ import numpy as np
264- smooth_scales = (268+ 
265- smoothScalesOptional.to(torch.float32)269+ x_np = x.to(torch.float32).numpy()
270+ smooth_np = (
271+ smoothScalesOptional.to(torch.float32).numpy()
266 if smoothScalesOptional is not None272 if smoothScalesOptional is not None
267 else None273 else None
268 )274 )
269- x_scaled = x_f * smooth_scales if smooth_scales is not None else x_f275+ 
270- amax = torch.amax(276+ output_dtype_str = str(kwargs.get("output_dtypes", ["int8"])[0])
271- torch.abs(x_scaled).view(-1, x_scaled.shape[-1]), dim=-1, keepdim=True277+ quant_mode = "perToken"
272- )278+ is_symmetrical = True
273- scale = amax / 127.0279+ 
274- scale = torch.where(scale == 0, torch.ones_like(scale), scale)280+ if quant_mode == "perchannel":
275- quantized = torch.round(x_scaled / scale)281+ result = _dynamic_quant_perchannel(
276- quantized = quantized.clamp(-128, 127).to(torch.int8)282+ x_np,
277- return [quantized, scale]283+ smooth_np,
284+ None,
285+ 2,
286+ quant_mode,
287+ is_symmetrical,
288+ output_dtype_str,
289+ )
290+ else:
291+ result = _dynamic_quant_common(
292+ x_np,
293+ smooth_np,
294+ None,
295+ 2,
296+ quant_mode,
297+ is_symmetrical,
298+ output_dtype_str,
299+ )
300+ 
301+ y_torch = torch.from_numpy(np.array(result[0])).to(torch.int8)
302+ scale_torch = torch.from_numpy(np.array(result[1])).to(torch.float32)
303+ return [y_torch, scale_torch]
278 304 
279 305 
280def aclnn_dynamic_quant_v3_golden(306def aclnn_dynamic_quant_v3_golden(
@@ -291,45 +317,57 @@ def aclnn_dynamic_quant_v3_golden(
291):317):
292 """318 """
293 Aclnn golden for aclnnDynamicQuantV3.319 Aclnn golden for aclnnDynamicQuantV3.
320+ Parameters follow @aclnnDynamicQuantV3GetWorkspaceSize without workspaceSize & executor.
321+ All the input Tensors are torch.Tensor.
294 """322 """
323+ import numpy as np
324+ 
295 if hasattr(dstType, "item"):325 if hasattr(dstType, "item"):
296 dstType = dstType.item()326 dstType = dstType.item()
297 if hasattr(isSymmetrical, "item"):327 if hasattr(isSymmetrical, "item"):
298 isSymmetrical = bool(isSymmetrical.item())328 isSymmetrical = bool(isSymmetrical.item())
299- if hasattr(quantMode, "item"):329+ if isinstance(quantMode, bytes):
330+ quantMode = quantMode.decode()
331+ elif hasattr(quantMode, "item"):
300 quantMode = quantMode.item()332 quantMode = quantMode.item()
301- x_f = x.to(torch.float32) if x.dtype != torch.float32 else x333+ 
302- smooth_scales = (334+ x_np = x.to(torch.float32).numpy()
303- smoothScalesOptional.to(torch.float32)335+ smooth_np = (
336+ smoothScalesOptional.to(torch.float32).numpy()
304 if smoothScalesOptional is not None337 if smoothScalesOptional is not None
305 else None338 else None
306 )339 )
307- x_scaled = x_f * smooth_scales if smooth_scales is not None else x_f340+ group_np = groupIndexOptional.numpy() if groupIndexOptional is not None else None
308- scale_max = 127.0341+ 
309- scale_max_no_sym = 255.0342+ output_dtype_str = str(kwargs.get("output_dtypes", ["int8"])[0])
310- offset = None343+ 
311- if not isSymmetrical:344+ if quantMode == "perchannel":
312- input_max = torch.max(x_scaled, dim=-1, keepdim=True).values345+ result = _dynamic_quant_perchannel(
313- input_min = torch.min(x_scaled, dim=-1, keepdim=True).values346+ x_np,
314- scale = (input_max - input_min) / scale_max_no_sym347+ smooth_np,
315- scale = torch.where(scale == 0, torch.ones_like(scale), scale)348+ group_np,
316- offset = scale_max - (input_max / scale)349+ dstType,
317- input_scaled = x_scaled / scale + offset350+ quantMode,
351+ isSymmetrical,
352+ output_dtype_str,
353+ )
318 else:354 else:
319- input_abs = torch.abs(x_scaled)355+ result = _dynamic_quant_common(
320- input_max = torch.max(input_abs, dim=-1, keepdim=True).values356+ x_np,
321- scale = input_max / scale_max357+ smooth_np,
322- scale = torch.where(scale == 0, torch.ones_like(scale), scale)358+ group_np,
323- input_scaled = x_scaled / scale359+ dstType,
324- round_data = torch.round(input_scaled)360+ quantMode if isinstance(quantMode, str) else "pertoken",
361+ isSymmetrical,
362+ output_dtype_str,
363+ )
364+ 
365+ y_torch = torch.from_numpy(np.array(result[0]))
325 if dstType == 2:366 if dstType == 2:
326- if isSymmetrical:367+ y_torch = y_torch.to(torch.int8)
327- round_data = round_data.clamp(-128, 127).to(torch.int8)368+ scale_torch = torch.from_numpy(np.array(result[1])).to(torch.float32)
328- else:369+ if len(result) > 2 and result[2] is not None:
329- round_data = round_data.clamp(0, 255).to(torch.uint8)370+ offset_torch = torch.from_numpy(np.array(result[2])).to(torch.float32)
330- scale_out = scale.squeeze(-1)
331- if offset is not None:
332- offset_out = offset.squeeze(-1)
333 else:371 else:
334- offset_out = torch.zeros_like(scale_out)372+ offset_torch = torch.zeros_like(scale_torch)
335- return [round_data, scale_out, offset_out]373+ return [y_torch, scale_torch, offset_torch]
@@ -12,25 +12,56 @@
12 12 
13import numpy as np13import numpy as np
14 14 
15-__input__ = {"kernel": {"dynamic_quant": "dynamic_quant_input"}}15+__input__ = {
16+ "kernel": {"dynamic_quant": "dynamic_quant_input"},
17+ "aclnn": {
18+ "aclnnDynamicQuant": "aclnn_dynamic_quant_input",
19+ "aclnnDynamicQuantV3": "aclnn_dynamic_quant_v3_input",
20+ },
21+}
16 22 
17 23 
18def dynamic_quant_input(x, smooth_scales=None, group_index=None, **kwargs):24def dynamic_quant_input(x, smooth_scales=None, group_index=None, **kwargs):
19 """25 """
20- Input function for dynamic_quant.26+ Input function for dynamic_quant (kernel).
21- All the parameters (names and order) follow @dynamic_quant_def.cpp without outputs.
22 All the input Tensors are numpy.ndarray.27 All the input Tensors are numpy.ndarray.
23- 
24- Returns:
25- List of input tensors (length must match Input count in _def.cpp)
26 """28 """
27 if group_index is not None:29 if group_index is not None:
28 S = np.prod(x.shape[:-1])30 S = np.prod(x.shape[:-1])
29 E = group_index.shape[0]31 E = group_index.shape[0]
30- 
31 group_index = np.random.choice(np.arange(1, S + 1), size=E, replace=False)32 group_index = np.random.choice(np.arange(1, S + 1), size=E, replace=False)
32 group_index.sort()33 group_index.sort()
33 group_index[-1] = S34 group_index[-1] = S
34 group_index = group_index.astype("int32")35 group_index = group_index.astype("int32")
35- 36+ return [x, smooth_scales, group_index]
37+ 
38+ 
39+def aclnn_dynamic_quant_input(*args, **kwargs):
40+ """
41+ Input function for aclnnDynamicQuant.
42+ TTK passes all tensor args; we only process the first two.
43+ """
44+ x = args[0] if len(args) > 0 else kwargs.get("x")
45+ smooth_scales = args[1] if len(args) > 1 else kwargs.get("smoothScalesOptional")
46+ return [x, smooth_scales]
47+ 
48+ 
49+def aclnn_dynamic_quant_v3_input(*args, **kwargs):
50+ """
51+ Input function for aclnnDynamicQuantV3.
52+ TTK passes all tensor args; we need to process group_index (3rd arg).
53+ """
54+ import torch
55+ 
56+ x = args[0] if len(args) > 0 else kwargs.get("x")
57+ smooth_scales = args[1] if len(args) > 1 else kwargs.get("smoothScalesOptional")
58+ group_index = args[2] if len(args) > 2 else kwargs.get("groupIndexOptional")
59+ 
60+ if group_index is not None:
61+ S = np.prod(x.shape[:-1])
62+ E = group_index.shape[0]
63+ gi = np.random.choice(np.arange(1, S + 1), size=E, replace=False)
64+ gi.sort()
65+ gi[-1] = S
66+ group_index = torch.from_numpy(gi.astype("int32"))
36 return [x, smooth_scales, group_index]67 return [x, smooth_scales, group_index]
@@ -0,0 +1,2 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,attributes,input_data_ranges,output_tensor_indexes,absolute_precision
2+aclnnDynamicQuantV3_float16_ND_fuzz_1,aclnnDynamicQuantV3,"('float16', 'float16', 'int32', 'int8', 'float32', 'float32')","('ND',)","((16, 71329),(1, 71329,),(1,),(16, 71329),(16,),(16,))","{'dstType': 2, 'isSymmetrical':False, 'quantMode':'pertoken'}","((-2, 2), (-2, 2), (16, 16))","(3, 4, 5)",0.0001
@@ -11,8 +11,14 @@
11# ----------------------------------------------------------------------------11# ----------------------------------------------------------------------------
12 12 
13import numpy as np13import numpy as np
14+import torch
14 15 
15-__golden__ = {"kernel": {"dynamic_quant_v2": "dynamic_quant_v2_golden"}}16+__golden__ = {
17+ "aclnn": {
18+ "aclnnDynamicQuantV2": "aclnn_dynamic_quant_v2_golden",
19+ },
20+ "kernel": {"dynamic_quant_v2": "dynamic_quant_v2_golden"},
21+}
16 22 
17 23 
18def _dynamic_quant_common(24def _dynamic_quant_common(
@@ -252,3 +258,68 @@ def dynamic_quant_v2_golden(
252 is_symmetrical,258 is_symmetrical,
253 output_dtype_str,259 output_dtype_str,
254 )260 )
261+ 
262+ 
263+def aclnn_dynamic_quant_v2_golden(
264+ x,
265+ smoothScalesOptional,
266+ groupIndexOptional,
267+ dstType,
268+ yOut=None,
269+ scaleOut=None,
270+ offsetOut=None,
271+ **kwargs,
272+):
273+ """
274+ Aclnn golden for aclnnDynamicQuantV2.
275+ Parameters follow @aclnnDynamicQuantV2GetWorkspaceSize without workspaceSize & executor.
276+ All the input Tensors are torch.Tensor.
277+ """
278+ import numpy as np
279+ 
280+ if hasattr(dstType, "item"):
281+ dstType = dstType.item()
282+ 
283+ x_np = x.to(torch.float32).numpy()
284+ smooth_np = (
285+ smoothScalesOptional.to(torch.float32).numpy()
286+ if smoothScalesOptional is not None
287+ else None
288+ )
289+ group_np = groupIndexOptional.numpy() if groupIndexOptional is not None else None
290+ 
291+ output_dtype_str = str(kwargs.get("output_dtypes", ["int8"])[0])
292+ 
293+ quant_mode = "pertoken"
294+ is_symmetrical = False
295+ 
296+ if quant_mode == "perchannel":
297+ result = _dynamic_quant_perchannel(
298+ x_np,
299+ smooth_np,
300+ group_np,
301+ dstType,
302+ quant_mode,
303+ is_symmetrical,
304+ output_dtype_str,
305+ )
306+ else:
307+ result = _dynamic_quant_common(
308+ x_np,
309+ smooth_np,
310+ group_np,
311+ dstType,
312+ quant_mode,
313+ is_symmetrical,
314+ output_dtype_str,
315+ )
316+ 
317+ y_torch = torch.from_numpy(np.array(result[0]))
318+ if dstType == 2:
319+ y_torch = y_torch.to(torch.int8)
320+ scale_torch = torch.from_numpy(np.array(result[1])).to(torch.float32)
321+ if len(result) > 2 and result[2] is not None:
322+ offset_torch = torch.from_numpy(np.array(result[2])).to(torch.float32)
323+ else:
324+ offset_torch = torch.zeros_like(scale_torch)
325+ return [y_torch, scale_torch, offset_torch]
@@ -12,25 +12,44 @@
12 12 
13import numpy as np13import numpy as np
14 14 
15-__input__ = {"kernel": {"dynamic_quant_v2": "dynamic_quant_v2_input"}}15+__input__ = {
16+ "kernel": {"dynamic_quant_v2": "dynamic_quant_v2_input"},
17+ "aclnn": {"aclnnDynamicQuantV2": "aclnn_dynamic_quant_v2_input"},
18+}
16 19 
17 20 
18def dynamic_quant_v2_input(x, smooth_scales=None, group_index=None, **kwargs):21def dynamic_quant_v2_input(x, smooth_scales=None, group_index=None, **kwargs):
19 """22 """
20 Input function for dynamic_quant_v2.23 Input function for dynamic_quant_v2.
21- All the parameters (names and order) follow @dynamic_quant_v2_def.cpp without outputs.
22 All the input Tensors are numpy.ndarray.24 All the input Tensors are numpy.ndarray.
23- 
24- Returns:
25- List of input tensors (length must match Input count in _def.cpp)
26 """25 """
27 if group_index is not None:26 if group_index is not None:
28 S = np.prod(x.shape[:-1])27 S = np.prod(x.shape[:-1])
29 E = group_index.shape[0]28 E = group_index.shape[0]
30- 
31 group_index = np.random.choice(np.arange(1, S + 1), size=E, replace=False)29 group_index = np.random.choice(np.arange(1, S + 1), size=E, replace=False)
32 group_index.sort()30 group_index.sort()
33 group_index[-1] = S31 group_index[-1] = S
34 group_index = group_index.astype("int32")32 group_index = group_index.astype("int32")
35- 33+ return [x, smooth_scales, group_index]
34+ 
35+ 
36+def aclnn_dynamic_quant_v2_input(*args, **kwargs):
37+ """
38+ Input function for aclnnDynamicQuantV2.
39+ All the input Tensors are torch.Tensor.
40+ TTK passes all tensor args; we only need to process group_index (3rd arg).
41+ """
42+ import torch
43+ 
44+ x = args[0] if len(args) > 0 else kwargs.get("x")
45+ smooth_scales = args[1] if len(args) > 1 else kwargs.get("smoothScalesOptional")
46+ group_index = args[2] if len(args) > 2 else kwargs.get("groupIndexOptional")
47+ 
48+ if group_index is not None:
49+ S = np.prod(x.shape[:-1])
50+ E = group_index.shape[0]
51+ gi = np.random.choice(np.arange(1, S + 1), size=E, replace=False)
52+ gi.sort()
53+ gi[-1] = S
54+ group_index = torch.from_numpy(gi.astype("int32"))
36 return [x, smooth_scales, group_index]55 return [x, smooth_scales, group_index]
@@ -0,0 +1,2 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,attributes,input_data_ranges,output_tensor_indexes,absolute_precision
2+aclnnDynamicQuantV2_float16_ND_fuzz_1,aclnnDynamicQuantV2,"('float16', 'float16', 'int32', 'int8', 'float32', 'float32')","('ND',)","((16, 71329),(1, 71329,),(1,),(16, 71329),(16,),(16,))",{'dstType': 2},"((-2, 2), (-2, 2), (16, 16))","(3, 4, 5)",0.0001