已合并
feat: Add aclnn st cases and golden functions #5200
yanzhi2024创建于 8月29日
feat: Add aclnn st cases and golden functions #5200
已合并
yanzhi2024创建于 8月29日
共 66 个文件变更+1196-291
@@ -12,25 +12,32 @@
12 12 
13import numpy13import numpy
14 14 
15+import torch
16+ 
15__golden__ = {17__golden__ = {
16- "kernel": {18+ "aclnn": {
17- "concat_d": "concat_d_golden"19+ "aclnnCat": "aclnn_cat_golden",
18- }20+ },
21+ "kernel": {"concat_d": "concat_d_golden"},
19}22}
20 23 
21 24 
22-def update_axis_for_hw_inner_format(ori_shape, axis, input_format, ori_format, reduce_mode=False):25+def update_axis_for_hw_inner_format(
26+ ori_shape, axis, input_format, ori_format, reduce_mode=False
27+):
23 if input_format in ("NDC1HWC0", "NC1HWC0"):28 if input_format in ("NDC1HWC0", "NC1HWC0"):
24 ori_shape_len = len(ori_shape) if -2 not in ori_shape else len(ori_format)29 ori_shape_len = len(ori_shape) if -2 not in ori_shape else len(ori_format)
25 axis = axis % ori_shape_len30 axis = axis % ori_shape_len
26 offset_6hd = 1 if input_format == "NDC1HWC0" else 031 offset_6hd = 1 if input_format == "NDC1HWC0" else 0
27- format_c_axis = 1 + offset_6hd if not reduce_mode else [1 + offset_6hd, 4 + offset_6hd]32+ format_c_axis = (
33+ 1 + offset_6hd if not reduce_mode else [1 + offset_6hd, 4 + offset_6hd]
34+ )
28 format_axis_map = {35 format_axis_map = {
29 "N": 0,36 "N": 0,
30 "C": format_c_axis,37 "C": format_c_axis,
31 "H": 2 + offset_6hd,38 "H": 2 + offset_6hd,
32 "W": 3 + offset_6hd,39 "W": 3 + offset_6hd,
33- "D": 140+ "D": 1,
34 }41 }
35 concat_dim_name = ori_format[axis]42 concat_dim_name = ori_format[axis]
36 axis = format_axis_map[concat_dim_name]43 axis = format_axis_map[concat_dim_name]
@@ -38,21 +45,33 @@ def update_axis_for_hw_inner_format(ori_shape, axis, input_format, ori_format, r
38 if input_format in ("FRACTAL_NZ",):45 if input_format in ("FRACTAL_NZ",):
39 axis = axis % len(ori_shape)46 axis = axis % len(ori_shape)
40 if axis == len(ori_shape) - 1:47 if axis == len(ori_shape) - 1:
41- axis = len(ori_shape) - 2 if not reduce_mode else [len(ori_shape) - 2, len(ori_shape) + 1]48+ axis = (
49+ len(ori_shape) - 2
50+ if not reduce_mode
51+ else [len(ori_shape) - 2, len(ori_shape) + 1]
52+ )
42 elif axis == len(ori_shape) - 2:53 elif axis == len(ori_shape) - 2:
43- axis = len(ori_shape) - 1 if not reduce_mode else [len(ori_shape) - 1, len(ori_shape) + 0]54+ axis = (
55+ len(ori_shape) - 1
56+ if not reduce_mode
57+ else [len(ori_shape) - 1, len(ori_shape) + 0]
58+ )
44 59 
45 if input_format in ("FRACTAL_Z", "FRACTAL_Z_3D"):60 if input_format in ("FRACTAL_Z", "FRACTAL_Z_3D"):
46 axis = axis % len(ori_shape)61 axis = axis % len(ori_shape)
47 offset_3d = 1 if input_format == "FRACTAL_Z_3D" else 062 offset_3d = 1 if input_format == "FRACTAL_Z_3D" else 0
48- format_c_axis = 0 + offset_3d if not reduce_mode else [0 + offset_3d, 5 + offset_3d]63+ format_c_axis = (
49- format_n_axis = 3 + offset_3d if not reduce_mode else [3 + offset_3d, 4 + offset_3d]64+ 0 + offset_3d if not reduce_mode else [0 + offset_3d, 5 + offset_3d]
65+ )
66+ format_n_axis = (
67+ 3 + offset_3d if not reduce_mode else [3 + offset_3d, 4 + offset_3d]
68+ )
50 format_axis_map = {69 format_axis_map = {
51 "N": format_n_axis,70 "N": format_n_axis,
52 "C": format_c_axis,71 "C": format_c_axis,
53 "H": 1 + offset_3d,72 "H": 1 + offset_3d,
54 "W": 2 + offset_3d,73 "W": 2 + offset_3d,
55- "D": 074+ "D": 0,
56 }75 }
57 concat_dim_name = ori_format[axis]76 concat_dim_name = ori_format[axis]
58 axis = format_axis_map[concat_dim_name]77 axis = format_axis_map[concat_dim_name]
@@ -61,7 +80,7 @@ def update_axis_for_hw_inner_format(ori_shape, axis, input_format, ori_format, r
61 80 
62 81 
63def concat_d_golden(x, *, concat_dim, N=1, **kwargs):82def concat_d_golden(x, *, concat_dim, N=1, **kwargs):
64- '''83+ """
65 Golden function for concat.84 Golden function for concat.
66 All the parameters (names and order) follow @concat_d_def.cpp without outputs.85 All the parameters (names and order) follow @concat_d_def.cpp without outputs.
67 All the input Tensors are numpy.ndarray.86 All the input Tensors are numpy.ndarray.
@@ -72,12 +91,20 @@ def concat_d_golden(x, *, concat_dim, N=1, **kwargs):
72 91 
73 Returns:92 Returns:
74 Output tensor93 Output tensor
75- '''94+ """
76 x_arrays = list(x)95 x_arrays = list(x)
77 96 
78- ori_shape = kwargs.get('input_ori_shapes', [x[0].shape])[0]97+ ori_shape = kwargs.get("input_ori_shapes", [x[0].shape])[0]
79- input_formats = kwargs.get('input_formats', ['ND'])98+ input_formats = kwargs.get("input_formats", ["ND"])
80- input_ori_formats = kwargs.get('input_ori_formats', ['ND'])99+ input_ori_formats = kwargs.get("input_ori_formats", ["ND"])
81- 100+ 
82- concat_dim = update_axis_for_hw_inner_format(ori_shape, concat_dim, input_formats[0], input_ori_formats[0])101+ concat_dim = update_axis_for_hw_inner_format(
102+ ori_shape, concat_dim, input_formats[0], input_ori_formats[0]
103+ )
83 return numpy.concatenate(x_arrays, axis=concat_dim)104 return numpy.concatenate(x_arrays, axis=concat_dim)
105+ 
106+ 
107+def aclnn_cat_golden(tensors, dim, out, **kwargs):
108+ if hasattr(dim, "item"):
109+ dim = dim.item()
110+ return [torch.cat(tensors, dim=dim)]
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_view_shapes,tensor_dtypes,attributes,output_tensor_indexes,input_data_ranges,precision_tolerances,absolute_precision
2+aclnnCat_float,aclnnCat,"(((3, 3), (3, 2)), (3, 5))","(('float32', 'float32'), 'float32')",{'dim': -1},"(1,)",,,
3+aclnnCat_bf16,aclnnCat,"(((4, 5), (4, 5)), (4, 10))","(('bfloat16', 'bfloat16'), 'bfloat16')",{'dim': -1},"(1,)",,,
4+aclnnCat_int8,aclnnCat,"(((13, 20), (13, 20)), (26, 20))","(('int8', 'int8'), 'int8')",{'dim': 0},"(1,)",,,
5+aclnnCat_int64,aclnnCat,"(((30, 30), (30, 30)), (30, 60))","(('int64', 'int64'), 'int64')",{'dim': -1},"(1,)",,,
6+aclnnCat_float16,aclnnCat,"(((3, 4), (3, 4)), (3, 8))","(('float16', 'float16'), 'float16')",{'dim': -1},"(1,)",,,
@@ -10,23 +10,30 @@
10# See LICENSE in the root of the software repository for the full text of the License.10# See LICENSE in the root of the software repository for the full text of the License.
11# ----------------------------------------------------------------------------11# ----------------------------------------------------------------------------
12import numpy as np12import numpy as np
13+import torch
13 14 
14__golden__ = {15__golden__ = {
15- "kernel": {16+ "aclnn": {
16- "pack": "pack_golden"17+ "aclnnStack": "aclnn_stack_golden",
17- }18+ },
19+ "kernel": {"pack": "pack_golden"},
18}20}
19- 21+ 
20-def pack_golden(x,22+ 
21- axis: int=0, N: int=1,23+def pack_golden(x, axis: int = 0, N: int = 1, **kwargs):
22- **kwargs):24+ """
23- '''
24 Kernel golden for pack.25 Kernel golden for pack.
25 All the parameters follow @pack_def.cpp without outputs.26 All the parameters follow @pack_def.cpp without outputs.
26 All the input Tensors are numpy.ndarray.27 All the input Tensors are numpy.ndarray.
27- kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, 28+ kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes,
28- input_formats, output_formats, input_ori_formats, output_ori_formats,29+ input_formats, output_formats, input_ori_formats, output_ori_formats,
29- input_dtypes, output_dtypes.30+ input_dtypes, output_dtypes.
30- '''31+ """
31- 32+ 
32 return np.stack(x, axis=axis)33 return np.stack(x, axis=axis)
34+ 
35+ 
36+def aclnn_stack_golden(tensors, dim, out, **kwargs):
37+ if hasattr(dim, "item"):
38+ dim = dim.item()
39+ return [torch.stack(tensors, dim=dim)]
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_view_shapes,tensor_dtypes,attributes,output_tensor_indexes,input_data_ranges,precision_tolerances,absolute_precision
2+aclnnStack_float,aclnnStack,"(((3, 3), (3, 3)), (3, 3, 2))","(('float32', 'float32'), 'float32')",{'dim': -1},"(1,)",,,
3+aclnnStack_bf16,aclnnStack,"(((4, 5), (4, 5)), (4, 5, 2))","(('bfloat16', 'bfloat16'), 'bfloat16')",{'dim': -1},"(1,)",,,
4+aclnnStack_int8,aclnnStack,"(((13, 20), (13, 20)), (2, 13, 20))","(('int8', 'int8'), 'int8')",{'dim': 0},"(1,)",,,
5+aclnnStack_int64,aclnnStack,"(((30, 30), (30, 30)), (30, 30, 2))","(('int64', 'int64'), 'int64')",{'dim': -1},"(1,)",,,
6+aclnnStack_float16,aclnnStack,"(((3, 4), (3, 4)), (3, 4, 2))","(('float16', 'float16'), 'float16')",{'dim': -1},"(1,)",,,
@@ -11,23 +11,33 @@
11# ----------------------------------------------------------------------------11# ----------------------------------------------------------------------------
12 12 
13import numpy as np13import numpy as np
14+import torch
14 15 
15 16 
16__golden__ = {17__golden__ = {
17- "kernel": {18+ "aclnn": {
18- "transpose": "transpose_golden"19+ "aclnnPermute": "aclnn_permute_golden",
19- }20+ },
21+ "kernel": {"transpose": "transpose_golden"},
20}22}
21 23 
22 24 
23def transpose_golden(x, perm, **kwargs):25def transpose_golden(x, perm, **kwargs):
24- '''26+ """
25 Kernel golden for transpose / transpose_d.27 Kernel golden for transpose / transpose_d.
26 All the parameters follow @transpose_def.cpp without outputs.28 All the parameters follow @transpose_def.cpp without outputs.
27 All the input Tensors are numpy.ndarray.29 All the input Tensors are numpy.ndarray.
28 kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes,30 kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes,
29 input_formats, output_formats, input_ori_formats, output_ori_formats,31 input_formats, output_formats, input_ori_formats, output_ori_formats,
30 input_dtypes, output_dtypes.32 input_dtypes, output_dtypes.
31- '''33+ """
32 perm_val = perm.tolist() if isinstance(perm, np.ndarray) else perm34 perm_val = perm.tolist() if isinstance(perm, np.ndarray) else perm
33 return np.transpose(x, perm_val)35 return np.transpose(x, perm_val)
36+ 
37+ 
38+def aclnn_permute_golden(self, dims=0, out=None, **kwargs):
39+ if hasattr(dims, "tolist"):
40+ dims = dims.tolist()
41+ elif isinstance(dims, int):
42+ dims = [dims]
43+ return [torch.permute(self, dims)]
@@ -0,0 +1,2 @@
1+testcase_name,api_name,tensor_dtypes,attributes,tensor_view_shapes,output_tensor_indexes,input_data_ranges,precision_tolerances,absolute_precision
2+aclnnPermute_00,aclnnPermute,"('float16', 'float16')","{'dims': [2,1,0]}","((256,64,128),(128,64,256))","(1,)","[[0, 0.001]]","((0.001, 0.001),)",0.000001
@@ -0,0 +1,22 @@
1+#!/usr/bin/env python3
2+# -*- coding: UTF-8 -*-
3+# ----------------------------------------------------------------------------
4+# Copyright (c) 2026 Huawei Technologies Co., Ltd.
5+# This program is free software, you can redistribute it and/or modify it under the terms and conditions of
6+# CANN Open Software License Agreement Version 2.0 (the "License").
7+# Please refer to the License for details. You may not use this file except in compliance with the License.
8+# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
9+# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
10+# See LICENSE in the root of the software repository for the full text of the License.
11+# ----------------------------------------------------------------------------
12+ 
13+__golden__ = {
14+ "aclnn": {
15+ "aclnnInplaceCopy": "aclnn_inplace_copy_golden",
16+ }
17+}
18+ 
19+ 
20+def aclnn_inplace_copy_golden(selfRef, src, **kwargs):
21+ selfRef.copy_(src)
22+ return [selfRef]
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,attributes,tensor_view_shapes,output_tensor_indexes,tensor_storage_shapes,tensor_view_strides,tensor_view_offsets,input_data_ranges
2+aclnnInplaceCopy_random_000001,aclnnInplaceCopy,"('float32', 'float32')","('ND', 'ND')","{'kernelShape': [2,2], 'strides': [2,2], 'autoPad' :0, 'pads' : [0], 'dilations': [1], 'ceilMode':False}","((4,2), (4,2))","(0,)","((4,8), (4,2))","((8,1), (2,1))","(0, 0,)","((0.001, 0.01),)"
3+aclnnInplaceCopy_random_000002,aclnnInplaceCopy,"('float16', 'float16')","('ND', 'ND')","{'kernelShape': [2,2], 'strides': [2,2], 'autoPad' :0, 'pads' : [0], 'dilations': [1], 'ceilMode':False}","((2,10), (2,10))","(0,)","((2,64), (2,10))","((64,1), (10,1))","(0, 0,)","((-1000, -10),)"
4+aclnnInplaceCopy_random_000003,aclnnInplaceCopy,"('bfloat16', 'bfloat16')","('ND', 'ND')","{'kernelShape': [2,2], 'strides': [2,2], 'autoPad' :0, 'pads' : [0], 'dilations': [1], 'ceilMode':False}","((6,40), (6,40))","(0,)","((6,64), (6,40))","((64,1), (40,1))","(0, 0,)","((1, 2),)"
5+aclnnInplaceCopy_random_000004,aclnnInplaceCopy,"('float32', 'float32')","('ND', 'ND')","{'kernelShape': [2,2], 'strides': [2,2], 'autoPad' :0, 'pads' : [0], 'dilations': [1], 'ceilMode':False}","((2, 3, 5), (2, 3, 5))","(0,)","((2, 5, 5), (2, 3, 5))","((25, 5, 1), (15, 5, 1))","(0, 0,)","((-1, -0.01),)"
6+aclnnInplaceCopy_random_000005,aclnnInplaceCopy,"('float32', 'float32')","('ND', 'ND')","{'kernelShape': [3], 'strides': [3], 'autoPad' :0, 'pads' : [1], 'dilations': [1], 'ceilMode':True}","((4, 5, 6), (4, 5, 6))","(0,)","((4, 10, 6), (4, 5, 6))","((60, 6, 1), (30, 6, 1))","(0, 0,)","((-0.01, 0.01),)"
@@ -41,7 +41,7 @@ def abs_golden(x, **kwargs):
41 return np.abs(x)41 return np.abs(x)
42 42 
43 43 
44-def aclnn_abs_golden(selfT, out=None, **kwargs):44+def aclnn_abs_golden(self, out=None, **kwargs):
45 """45 """
46 Aclnn golden for aclnnAbs.46 Aclnn golden for aclnnAbs.
47 Parameters follow @aclnnAbsGetWorkspaceSize without workspaceSize & executor.47 Parameters follow @aclnnAbsGetWorkspaceSize without workspaceSize & executor.
@@ -50,4 +50,4 @@ def aclnn_abs_golden(selfT, out=None, **kwargs):
50 kwargs may contain: tensor_dtypes, tensor_formats, scalar_dtypes,50 kwargs may contain: tensor_dtypes, tensor_formats, scalar_dtypes,
51 use_torch, short_soc_version, testcase_name.51 use_torch, short_soc_version, testcase_name.
52 """52 """
53- return torch.abs(selfT)53+ return torch.abs(self)
@@ -1,14 +1,6 @@
1-testcase_name,api_name,tensor_view_shapes,tensor_dtypes,tensor_formats,attributes,output_tensor_indexes,input_data_ranges,absolute_precision1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,absolute_precision
2-Abs_float32_ND_fuzz_1,aclnnAbs,"((196, 2, 1, 76, 1, 1), (196, 2, 1, 76, 1, 1))","('float32', 'float32')","('ND',)",{},"(-1,)","((-1000, -10),)",0.00012+Abs_float32_ND_fuzz_1,aclnnAbs,"('float32', 'float32')","('ND',)","((196, 2, 1, 76, 1, 1), (196, 2, 1, 76, 1, 1))","((-1000, -10),)","(-1,)",0.0001
3-Abs_float16_ND_fuzz_2,aclnnAbs,"((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","('float16', 'float16')","('ND',)",{},"(-1,)","((0, 0),)",0.0013+Abs_float16_ND_fuzz_2,aclnnAbs,"('float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)",0.001
4-Abs_float32_ND_fuzz_3,aclnnAbs,"((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","('float32', 'float32')","('ND',)",{},"(-1,)","((-3.4e+38, 3.4e+38),)",0.00014+Abs_float32_ND_fuzz_3,aclnnAbs,"('float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)",0.0001
5-Abs_float16_ND_fuzz_4,aclnnAbs,"((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","('float16', 'float16')","('ND',)",{},"(-1,)","((-10, -2),)",0.0015+Abs_float16_ND_fuzz_4,aclnnAbs,"('float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)",0.001
6-Abs_float16_ND_fuzz_5,aclnnAbs,"((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","('float16', 'float16')","('ND',)",{},"(-1,)","((-1, 1),)",0.0016+Abs_float16_ND_fuzz_5,aclnnAbs,"('float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)",0.001
7-Abs_float32_ND_fuzz_6,aclnnAbs,"((1, 64, 10, 1, 192, 1), (1, 64, 10, 1, 192, 1))","('float32', 'float32')","('ND',)",{},"(-1,)","((-1, 1),)",0.0001
8-Abs_bfloat16_ND_fuzz_7,aclnnAbs,"((1, 64, 10, 1, 192, 1), (1, 64, 10, 1, 192, 1))","('bfloat16', 'bfloat16')","('ND',)",{},"(-1,)","((-1, 1),)",0.0001
9-Abs_int8_ND_fuzz_8,aclnnAbs,"((1, 64, 10, 1, 192, 1), (1, 64, 10, 1, 192, 1))","('int8', 'int8')","('ND',)",{},"(-1,)","((-1, 1),)",0.0001
10-Abs_int32_ND_fuzz_9,aclnnAbs,"((1, 64, 10, 1, 192, 1), (1, 64, 10, 1, 192, 1))","('int32', 'int32')","('ND',)",{},"(-1,)","((-1, 1),)",0.0001
11-Abs_int64_ND_fuzz_10,aclnnAbs,"((1, 64, 10, 1, 192, 1), (1, 64, 10, 1, 192, 1))","('int64', 'int64')","('ND',)",{},"(-1,)","((-1, 1),)",0.0001
12-Abs_int8_ND_fuzz_11,aclnnAbs,"((1, 64, 10, 1, 192, 1), (1, 64, 10, 1, 192, 1))","('int8', 'int8')","('ND',)",{},"(-1,)","((-1, 1),)",0.0001
13-Abs_int16_ND_fuzz_12,aclnnAbs,"((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","('int16', 'int16')","('ND',)",{},"(-1,)","((-1, 1),)",0.001
14- 
@@ -9,84 +9,98 @@
9# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.9# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
10# See LICENSE in the root of the software repository for the full text of the License.10# See LICENSE in the root of the software repository for the full text of the License.
11# ----------------------------------------------------------------------------11# ----------------------------------------------------------------------------
12- 12+ 
13import numpy as np13import numpy as np
14- 14+ 
15__golden__ = {15__golden__ = {
16- "kernel": {16+ "aclnn": {
17- "cast": "cast_golden"17+ "aclnnCast": "aclnn_cast_golden",
18- }18+ },
19+ "kernel": {"cast": "cast_golden"},
19}20}
20- 21+ 
21_DATA_TYPE_INT_TO_STR = {22_DATA_TYPE_INT_TO_STR = {
22- 0: 'float32',23+ 0: "float32",
23- 1: 'float16',24+ 1: "float16",
24- 2: 'int8',25+ 2: "int8",
25- 3: 'int32',26+ 3: "int32",
26- 4: 'uint8',27+ 4: "uint8",
27- 6: 'int16',28+ 6: "int16",
28- 7: 'uint16',29+ 7: "uint16",
29- 8: 'uint32',30+ 8: "uint32",
30- 9: 'int64',31+ 9: "int64",
31- 10: 'uint64',32+ 10: "uint64",
32- 11: 'double',33+ 11: "double",
33- 12: 'bool',34+ 12: "bool",
34- 16: 'complex64',35+ 16: "complex64",
35- 17: 'complex128',36+ 17: "complex128",
36- 27: 'bfloat16',37+ 27: "bfloat16",
37- 29: 'int4',38+ 29: "int4",
38- 30: 'uint1',39+ 30: "uint1",
39- 33: 'complex32',40+ 33: "complex32",
40- 34: 'hifloat8',41+ 34: "hifloat8",
41- 35: 'float8_e5m2',42+ 35: "float8_e5m2",
42- 36: 'float8_e4m3fn',43+ 36: "float8_e4m3fn",
43- 40: 'float4_e2m1',44+ 40: "float4_e2m1",
44- 41: 'float4_e1m2',45+ 41: "float4_e1m2",
45}46}
46- 47+ 
47-_SPECIAL_DTYPES = ("bfloat16", "int4",48+_SPECIAL_DTYPES = (
48- "float8_e5m2", "float8_e4m3fn",49+ "bfloat16",
49- "float4_e2m1", "float4_e1m2",50+ "int4",
50- "hifloat8")51+ "float8_e5m2",
51- 52+ "float8_e4m3fn",
53+ "float4_e2m1",
54+ "float4_e1m2",
55+ "hifloat8",
56+)
57+ 
58+ 
52def _resolve_custom_numpy_dtype(dtype_str):59def _resolve_custom_numpy_dtype(dtype_str):
53 if dtype_str == "bfloat16":60 if dtype_str == "bfloat16":
54 from ml_dtypes import bfloat1661 from ml_dtypes import bfloat16
62+ 
55 return bfloat1663 return bfloat16
56 elif dtype_str == "int4":64 elif dtype_str == "int4":
57 from ml_dtypes import int465 from ml_dtypes import int4
66+ 
58 return int467 return int4
59 elif dtype_str == "float8_e5m2":68 elif dtype_str == "float8_e5m2":
60 from ml_dtypes import float8_e5m269 from ml_dtypes import float8_e5m2
70+ 
61 return float8_e5m271 return float8_e5m2
62 elif dtype_str == "float8_e4m3fn":72 elif dtype_str == "float8_e4m3fn":
63 from ml_dtypes import float8_e4m3fn73 from ml_dtypes import float8_e4m3fn
74+ 
64 return float8_e4m3fn75 return float8_e4m3fn
65 elif dtype_str == "hifloat8":76 elif dtype_str == "hifloat8":
66 from en_dtypes import hifloat877 from en_dtypes import hifloat8
78+ 
67 return hifloat879 return hifloat8
68 elif dtype_str == "float4_e2m1":80 elif dtype_str == "float4_e2m1":
69 from ml_dtypes import float4_e2m181 from ml_dtypes import float4_e2m1
82+ 
70 return float4_e2m183 return float4_e2m1
71 elif dtype_str == "float4_e1m2":84 elif dtype_str == "float4_e1m2":
72 from ml_dtypes import float4_e1m285 from ml_dtypes import float4_e1m2
86+ 
73 return float4_e1m287 return float4_e1m2
74 return None88 return None
75- 89+ 
76-def cast_golden(x,90+ 
77- dst_type: int,91+def cast_golden(x, dst_type: int, **kwargs):
78- **kwargs):92+ """
79- '''
80 Kernel golden for cast.93 Kernel golden for cast.
81 All the parameters follow @cast_def.cpp without outputs.94 All the parameters follow @cast_def.cpp without outputs.
82 All the input Tensors are numpy.ndarray.95 All the input Tensors are numpy.ndarray.
83 kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes,96 kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes,
84- input_formats, output_formats, input_ori_formats, output_ori_formats,97+ input_formats, output_formats, input_ori_formats, output_ori_formats,
85- input_dtypes, output_dtypes.98+ input_dtypes, output_dtypes.
86- '''99+ """
87 dst_type_str = _DATA_TYPE_INT_TO_STR.get(dst_type, str(dst_type))100 dst_type_str = _DATA_TYPE_INT_TO_STR.get(dst_type, str(dst_type))
88- if (x.dtype.name == "bfloat16" and dst_type_str == "hifloat8") or \101+ if (x.dtype.name == "bfloat16" and dst_type_str == "hifloat8") or (
89- (x.dtype.name == "hifloat8" and dst_type_str == "bfloat16"):102+ x.dtype.name == "hifloat8" and dst_type_str == "bfloat16"
103+ ):
90 np_dtype = _resolve_custom_numpy_dtype(dst_type_str)104 np_dtype = _resolve_custom_numpy_dtype(dst_type_str)
91 return x.astype(np.float32).astype(np_dtype)105 return x.astype(np.float32).astype(np_dtype)
92 elif dst_type_str in _SPECIAL_DTYPES:106 elif dst_type_str in _SPECIAL_DTYPES:
@@ -102,3 +116,15 @@ def cast_golden(x,
102 return x.astype(np.bool_)116 return x.astype(np.bool_)
103 else:117 else:
104 return x.astype(getattr(np, dst_type_str))118 return x.astype(getattr(np, dst_type_str))
119+ 
120+ 
121+def aclnn_cast_golden(self, dtype=0, out=None, **kwargs):
122+ """
123+ Aclnn golden for aclnnCast.
124+ Parameters follow @aclnnCastGetWorkspaceSize without workspaceSize & executor.
125+ All the input Tensors are torch.Tensor.
126+ """
127+ from ttk.utilities import acl_to_torch_dtype
128+ 
129+ torch_dtype = acl_to_torch_dtype([dtype])[0]
130+ return self.to(dtype=torch_dtype)
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,attributes,absolute_precision
2+cast_fuzz_069,aclnnCast,"('uint8','float16')","('ND',)","((24, 16),(24, 16))","((0, 255),)","(-1,)",{'dtype': 1},0.0001
3+cast_fuzz_049,aclnnCast,"('int64','int32')","('ND',)","((3, 45),(3, 45))","((-9223372036854775808, 9223372036854775807),)","(-1,)",{'dtype': 3},0.0001
4+cast_fuzz_059,aclnnCast,"('int16','int8')","('ND',)","((1,), (1,))","((-32768, 32767),)","(-1,)",{'dtype': 2},0.001
5+cast_fuzz_050,aclnnCast,"('int64','int16')","('ND',)","((25, 1, 32),(25, 1, 32))","((-9223372036854775808, 9223372036854775807),)","(-1,)",{'dtype': 6},0.001
6+cast_fuzz_003,aclnnCast,"('bfloat16','float32')","('ND',)","((14, 17),(14, 17))","((-3.389531389251505e+38, 3.389531389251505e+38),)","(-1,)",{'dtype': 0},0.0001
@@ -9,23 +9,34 @@
9# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.9# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
10# See LICENSE in the root of the software repository for the full text of the License.10# See LICENSE in the root of the software repository for the full text of the License.
11# ----------------------------------------------------------------------------11# ----------------------------------------------------------------------------
12- 12+ 
13import numpy as np13import numpy as np
14- 14+import torch
15+ 
15__golden__ = {16__golden__ = {
16- "kernel": {17+ "aclnn": {
17- "ceil": "ceil_golden"18+ "aclnnCeil": "aclnn_ceil_golden",
18- }19+ },
20+ "kernel": {"ceil": "ceil_golden"},
19}21}
20- 22+ 
21-def ceil_golden(x,23+ 
22- **kwargs):24+def ceil_golden(x, **kwargs):
23- '''25+ """
24 Kernel golden for ceil.26 Kernel golden for ceil.
25 All the parameters follow @ceil_def.cpp without outputs.27 All the parameters follow @ceil_def.cpp without outputs.
26 All the input Tensors are numpy.ndarray.28 All the input Tensors are numpy.ndarray.
27 kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes,29 kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes,
28- input_formats, output_formats, input_ori_formats, output_ori_formats,30+ input_formats, output_formats, input_ori_formats, output_ori_formats,
29- input_dtypes, output_dtypes.31+ input_dtypes, output_dtypes.
30- '''32+ """
31 return np.ceil(x)33 return np.ceil(x)
34+ 
35+ 
36+def aclnn_ceil_golden(self, out=None, **kwargs):
37+ """
38+ Aclnn golden for aclnnCeil.
39+ Parameters follow @aclnnCeilGetWorkspaceSize without workspaceSize & executor.
40+ All the input Tensors are torch.Tensor.
41+ """
42+ return [torch.ceil(self)]
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,absolute_precision
2+Ceil_float32_ND_fuzz_1,aclnnCeil,"('float32', 'float32')","('ND',)","((196, 2, 1, 76),(196, 2, 1, 76))","((-1000, -10),)","(-1,)","('float32',)",0.0001
3+Ceil_float16_ND_fuzz_2,aclnnCeil,"('float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('float16',)",0.001
4+Ceil_float32_ND_fuzz_3,aclnnCeil,"('float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)","('float32',)",0.0001
5+Ceil_float16_ND_fuzz_4,aclnnCeil,"('float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('float16',)",0.001
6+Ceil_float16_ND_fuzz_5,aclnnCeil,"('float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)","('float16',)",0.001
@@ -1,4 +1,4 @@
1- #!/usr/bin/env python31+#!/usr/bin/env python3
2# -*- coding: UTF-8 -*-2# -*- coding: UTF-8 -*-
3# ----------------------------------------------------------------------------3# ----------------------------------------------------------------------------
4# Copyright (c) 2026 Huawei Technologies Co., Ltd.4# Copyright (c) 2026 Huawei Technologies Co., Ltd.
@@ -11,27 +11,38 @@
11# ----------------------------------------------------------------------------11# ----------------------------------------------------------------------------
12 12 
13import numpy as np13import numpy as np
14+import torch
14 15 
15 16 
16__golden__ = {17__golden__ = {
17- "kernel": {18+ "aclnn": {
18- "cos": "cos_golden"19+ "aclnnCos": "aclnn_cos_golden",
19- }20+ },
21+ "kernel": {"cos": "cos_golden"},
20}22}
21 23 
22 24 
23def cos_golden(x, **kwargs):25def cos_golden(x, **kwargs):
24- '''26+ """
25 Kernel golden for cos.27 Kernel golden for cos.
26 All the parameters follow @cos_def.cpp without outputs.28 All the parameters follow @cos_def.cpp without outputs.
27 All the input Tensors are numpy.ndarray.29 All the input Tensors are numpy.ndarray.
28 kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes,30 kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes,
29 input_formats, output_formats, input_ori_formats, output_ori_formats,31 input_formats, output_formats, input_ori_formats, output_ori_formats,
30 input_dtypes, output_dtypes.32 input_dtypes, output_dtypes.
31- '''33+ """
32 ori_dtype = x.dtype34 ori_dtype = x.dtype
33 if ori_dtype.name in ("float16", "bfloat16"):35 if ori_dtype.name in ("float16", "bfloat16"):
34 x_cast = x.astype(np.float32)36 x_cast = x.astype(np.float32)
35 res = np.cos(x_cast)37 res = np.cos(x_cast)
36 return res.astype(ori_dtype, copy=False)38 return res.astype(ori_dtype, copy=False)
37- return np.cos(x)39+ return np.cos(x)
40+ 
41+ 
42+def aclnn_cos_golden(input, out=None, **kwargs):
43+ """
44+ Aclnn golden for aclnnCos.
45+ Parameters follow @aclnnCosGetWorkspaceSize without workspaceSize & executor.
46+ All the input Tensors are torch.Tensor.
47+ """
48+ return [torch.cos(input)]
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,absolute_precision
2+Cos_float32_ND_fuzz_1,aclnnCos,"('float32', 'float32')","('ND',)","((196, 2, 1, 76),(196, 2, 1, 76))","((-1000, -10),)","(-1,)","('float32',)",0.0001
3+Cos_float16_ND_fuzz_2,aclnnCos,"('float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('float16',)",0.001
4+Cos_float32_ND_fuzz_3,aclnnCos,"('float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)","('float32',)",0.0001
5+Cos_float16_ND_fuzz_4,aclnnCos,"('float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('float16',)",0.001
6+Cos_float16_ND_fuzz_5,aclnnCos,"('float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)","('float16',)",0.001
@@ -14,26 +14,27 @@ import numpy as np
14 14 
15 15 
16__golden__ = {16__golden__ = {
17- "kernel": {17+ "aclnn": {
18- "erf": "erf_golden"18+ "aclnnErf": "aclnn_erf_golden",
19- }19+ },
20+ "kernel": {"erf": "erf_golden"},
20}21}
21 22 
22 23 
23def erf_golden(x, **kwargs):24def erf_golden(x, **kwargs):
24- '''25+ """
25 Kernel golden for erf.26 Kernel golden for erf.
26 All the parameters follow @erf_def.cpp without outputs.27 All the parameters follow @erf_def.cpp without outputs.
27 All the input Tensors are numpy.ndarray.28 All the input Tensors are numpy.ndarray.
28 kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes,29 kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes,
29 input_formats, output_formats, input_ori_formats, output_ori_formats,30 input_formats, output_formats, input_ori_formats, output_ori_formats,
30 input_dtypes, output_dtypes.31 input_dtypes, output_dtypes.
31- '''32+ """
32 import torch33 import torch
33- 34+ 
34 ori_dtype = kwargs.get("input_dtypes", ["float32"])[0]35 ori_dtype = kwargs.get("input_dtypes", ["float32"])[0]
35 x_dtype = x.dtype36 x_dtype = x.dtype
36- 37+ 
37 if ori_dtype and "bfloat16" in str(ori_dtype).lower():38 if ori_dtype and "bfloat16" in str(ori_dtype).lower():
38 x_tensor = torch.from_numpy(x.astype(np.float32))39 x_tensor = torch.from_numpy(x.astype(np.float32))
39 output = torch.erf(x_tensor)40 output = torch.erf(x_tensor)
@@ -45,4 +46,15 @@ def erf_golden(x, **kwargs):
45 else:46 else:
46 x_tensor = torch.from_numpy(x)47 x_tensor = torch.from_numpy(x)
47 output = torch.erf(x_tensor)48 output = torch.erf(x_tensor)
48- return output.numpy()49+ return output.numpy()
50+ 
51+ 
52+def aclnn_erf_golden(self, out=None, **kwargs):
53+ """
54+ Aclnn golden for aclnnErf.
55+ Parameters follow @aclnnErfGetWorkspaceSize without workspaceSize & executor.
56+ All the input Tensors are torch.Tensor.
57+ """
58+ import torch
59+ 
60+ return [torch.erf(self)]
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes
2+Erf_float32_ND_fuzz_1,aclnnErf,"('float32', 'float32')","('ND',)","((196, 2, 1, 76),(196, 2, 1, 76))","((-1000, -10),)","(-1,)","('float32',)"
3+Erf_float16_ND_fuzz_2,aclnnErf,"('float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('float16',)"
4+Erf_float32_ND_fuzz_3,aclnnErf,"('float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)","('float32',)"
5+Erf_float16_ND_fuzz_4,aclnnErf,"('float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('float16',)"
6+Erf_bfloat16_ND_fuzz_5,aclnnErf,"('bfloat16', 'bfloat16')","('ND',)","((1, 64, 10, 1, 192, 1), (1, 64, 10, 1, 192, 1))","((-1000, -10),)","(-1,)","('bfloat16',)"
@@ -12,22 +12,22 @@
12import numpy as np12import numpy as np
13 13 
14__golden__ = {14__golden__ = {
15- "kernel": {15+ "aclnn": {
16- "exp": "exp_golden"16+ "aclnnExp": "aclnn_exp_golden",
17- }17+ },
18+ "kernel": {"exp": "exp_golden"},
18}19}
19- 20+ 
20-def exp_golden(x,21+ 
21- base: float=-1.0, scale: float=1.0, shift: float=0.0,22+def exp_golden(x, base: float = -1.0, scale: float = 1.0, shift: float = 0.0, **kwargs):
22- **kwargs):23+ """
23- '''
24 Kernel golden for exp.24 Kernel golden for exp.
25 All the parameters follow @exp_def.cpp without outputs.25 All the parameters follow @exp_def.cpp without outputs.
26 All the input Tensors are numpy.ndarray.26 All the input Tensors are numpy.ndarray.
27- kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, 27+ kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes,
28- input_formats, output_formats, input_ori_formats, output_ori_formats,28+ input_formats, output_formats, input_ori_formats, output_ori_formats,
29- input_dtypes, output_dtypes.29+ input_dtypes, output_dtypes.
30- '''30+ """
31 import torch31 import torch
32 32 
33 x_dtype = x.dtype33 x_dtype = x.dtype
@@ -39,10 +39,21 @@ def exp_golden(x,
39 if base == -1:39 if base == -1:
40 output = torch.exp(x)40 output = torch.exp(x)
41 else:41 else:
42- output = torch.exp((scale * x + shift)*np.log(base))42+ output = torch.exp((scale * x + shift) * np.log(base))
43 elif base == -1:43 elif base == -1:
44 output = torch.exp(scale * x + shift)44 output = torch.exp(scale * x + shift)
45 else:45 else:
46- output = torch.exp((scale * x + shift)*np.log(base))46+ output = torch.exp((scale * x + shift) * np.log(base))
47 47 
48 return output.numpy().astype(x_dtype, copy=False)48 return output.numpy().astype(x_dtype, copy=False)
49+ 
50+ 
51+def aclnn_exp_golden(self, out=None, **kwargs):
52+ """
53+ Aclnn golden for aclnnExp.
54+ Parameters follow @aclnnExpGetWorkspaceSize without workspaceSize & executor.
55+ All the input Tensors are torch.Tensor.
56+ """
57+ import torch
58+ 
59+ return [torch.exp(self)]
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,absolute_precision
2+Exp_float32_ND_fuzz_1,aclnnExp,"('float32', 'float32')","('ND',)","((196, 2, 1, 76),(196, 2, 1, 76))","((-1000, -10),)","(-1,)","('float32',)",0.0001
3+Exp_float16_ND_fuzz_2,aclnnExp,"('float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('float16',)",0.001
4+Exp_float32_ND_fuzz_3,aclnnExp,"('float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)","('float32',)",0.0001
5+Exp_float16_ND_fuzz_4,aclnnExp,"('float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('float16',)",0.001
6+Exp_float16_ND_fuzz_5,aclnnExp,"('float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)","('float16',)",0.001
@@ -11,14 +11,25 @@
11# ----------------------------------------------------------------------------11# ----------------------------------------------------------------------------
12 12 
13import numpy as np13import numpy as np
14+import torch
14 15 
15 16 
16__golden__ = {17__golden__ = {
17- "kernel": {18+ "aclnn": {
18- "is_finite": "is_finite_golden"19+ "aclnnIsFinite": "aclnn_is_finite_golden",
19- }20+ },
21+ "kernel": {"is_finite": "is_finite_golden"},
20}22}
21 23 
22 24 
23def is_finite_golden(x, **kwargs):25def is_finite_golden(x, **kwargs):
24- return np.isfinite(x)26+ return np.isfinite(x)
27+ 
28+ 
29+def aclnn_is_finite_golden(self, out=None, **kwargs):
30+ """
31+ Aclnn golden for aclnnIsFinite.
32+ Parameters follow @aclnnIsFiniteGetWorkspaceSize without workspaceSize & executor.
33+ All the input Tensors are torch.Tensor.
34+ """
35+ return [torch.ops.aten.isfinite(self)]
@@ -0,0 +1,6 @@
1+,testcase_name,api_name,tensor_view_shapes,tensor_dtypes
2+0,aclnnIsFinite_00000,aclnnIsFinite,"[[38],[38]]","('float16','bool')"
3+1,aclnnIsFinite_00001,aclnnIsFinite,"[[47, 15, 8, 21, 5],[47, 15, 8, 21, 5]]","('float32','bool')"
4+2,aclnnIsFinite_00002,aclnnIsFinite,"[[10, 42, 4, 15],[10, 42, 4, 15]]","('float16','bool')"
5+3,aclnnIsFinite_00003,aclnnIsFinite,"[[43, 22],[43, 22]]","('float16','bool')"
6+4,aclnnIsFinite_00004,aclnnIsFinite,"[[9, 14, 29],[9, 14, 29]]","('bfloat16','bool')"
@@ -11,14 +11,20 @@
11# ----------------------------------------------------------------------------11# ----------------------------------------------------------------------------
12 12 
13import numpy as np13import numpy as np
14+import torch
14 15 
15 16 
16__golden__ = {17__golden__ = {
17- "kernel": {18+ "aclnn": {
18- "is_inf": "is_inf_golden"19+ "aclnnIsInf": "aclnn_is_inf_golden",
19- }20+ },
21+ "kernel": {"is_inf": "is_inf_golden"},
20}22}
21 23 
22 24 
23def is_inf_golden(x, **kwargs):25def is_inf_golden(x, **kwargs):
24- return np.isinf(x)26+ return np.isinf(x)
27+ 
28+ 
29+def aclnn_is_inf_golden(x, out=None, **kwargs):
30+ return [torch.isinf(x)]
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes
2+IsInf_float32_ND_fuzz_1,aclnnIsInf,"('float32', 'bool')","('ND',)","((196, 2, 1, 76),(196, 2, 1, 76))","((-1000, -10),)","(-1,)","('bool',)"
3+IsInf_float16_ND_fuzz_2,aclnnIsInf,"('float16', 'bool')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('bool',)"
4+IsInf_float32_ND_fuzz_3,aclnnIsInf,"('float32', 'bool')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)","('bool',)"
5+IsInf_float16_ND_fuzz_4,aclnnIsInf,"('float16', 'bool')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('bool',)"
6+IsInf_float16_ND_fuzz_5,aclnnIsInf,"('float16', 'bool')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((inf, inf),)","(-1,)","('bool',)"
@@ -13,28 +13,28 @@ import numpy as np
13import torch13import torch
14 14 
15__golden__ = {15__golden__ = {
16- "kernel": {16+ "aclnn": {
17- "log": "log_golden"17+ "aclnnLog": "aclnn_log_golden",
18- }18+ },
19+ "kernel": {"log": "log_golden"},
19}20}
20- 21+ 
21-def log_golden(x,22+ 
22- base: float=-1.0, scale: float=1.0, shift: float=0.0,23+def log_golden(x, base: float = -1.0, scale: float = 1.0, shift: float = 0.0, **kwargs):
23- **kwargs):24+ """
24- '''
25 Kernel golden for log.25 Kernel golden for log.
26 All the parameters follow @log_def.cpp without outputs.26 All the parameters follow @log_def.cpp without outputs.
27 All the input Tensors are numpy.ndarray.27 All the input Tensors are numpy.ndarray.
28- kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, 28+ kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes,
29- input_formats, output_formats, input_ori_formats, output_ori_formats,29+ input_formats, output_formats, input_ori_formats, output_ori_formats,
30- input_dtypes, output_dtypes.30+ input_dtypes, output_dtypes.
31- '''31+ """
32 x_dtype = x.dtype32 x_dtype = x.dtype
33 if x_dtype.name == "bfloat16" or x_dtype.name == "float16":33 if x_dtype.name == "bfloat16" or x_dtype.name == "float16":
34 x = torch.from_numpy(x.astype(np.float32))34 x = torch.from_numpy(x.astype(np.float32))
35- else :35+ else:
36 x = torch.from_numpy(x)36 x = torch.from_numpy(x)
37- 37+ 
38 if scale == 1 and shift == 0:38 if scale == 1 and shift == 0:
39 if base == -1:39 if base == -1:
40 output = torch.log(x)40 output = torch.log(x)
@@ -43,10 +43,21 @@ def log_golden(x,
43 elif base == 10:43 elif base == 10:
44 output = torch.log10(x)44 output = torch.log10(x)
45 else:45 else:
46- output = torch.log((scale * x + shift))/np.log(base)46+ output = torch.log((scale * x + shift)) / np.log(base)
47 elif base == -1:47 elif base == -1:
48 output = torch.log((scale * x + shift))48 output = torch.log((scale * x + shift))
49 else:49 else:
50- output = torch.log((scale * x + shift))/np.log(base)50+ output = torch.log((scale * x + shift)) / np.log(base)
51 51 
52 return output.numpy().astype(x_dtype, copy=False)52 return output.numpy().astype(x_dtype, copy=False)
53+ 
54+ 
55+def aclnn_log_golden(self, out=None, **kwargs):
56+ """
57+ Aclnn golden for aclnnLog.
58+ Parameters follow @aclnnLogGetWorkspaceSize without workspaceSize & executor.
59+ All the input Tensors are torch.Tensor.
60+ """
61+ import torch
62+ 
63+ return [torch.log(self)]
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,absolute_precision
2+Log_float32_ND_fuzz_1,aclnnLog,"('float32', 'float32')","('ND',)","((196, 2, 1, 76),(196, 2, 1, 76))","((-1000, -10),)","(-1,)","('float32',)",0.0001
3+Log_float16_ND_fuzz_2,aclnnLog,"('float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('float16',)",0.001
4+Log_float32_ND_fuzz_3,aclnnLog,"('float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)","('float32',)",0.0001
5+Log_float16_ND_fuzz_4,aclnnLog,"('float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('float16',)",0.001
6+Log_float16_ND_fuzz_5,aclnnLog,"('float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)","('float16',)",0.001
@@ -15,19 +15,31 @@ import torch
15 15 
16 16 
17__golden__ = {17__golden__ = {
18- "kernel": {18+ "aclnn": {
19- "log1p": "log1p_golden"19+ "aclnnLog1p": "aclnn_log1p_golden",
20- }20+ },
21+ "kernel": {"log1p": "log1p_golden"},
21}22}
22 23 
23 24 
24def log1p_golden(x, **kwargs):25def log1p_golden(x, **kwargs):
25 ori_dtype = kwargs.get("input_dtypes", ["float32"])[0]26 ori_dtype = kwargs.get("input_dtypes", ["float32"])[0]
26 x_dtype = x.dtype27 x_dtype = x.dtype
27- 28+ 
28 if "bfloat16" in str(ori_dtype).lower() or "float16" in str(ori_dtype).lower():29 if "bfloat16" in str(ori_dtype).lower() or "float16" in str(ori_dtype).lower():
29 x_tensor = torch.from_numpy(x.astype(np.float32))30 x_tensor = torch.from_numpy(x.astype(np.float32))
30 output = torch.log1p(x_tensor)31 output = torch.log1p(x_tensor)
31 return output.numpy().astype(x_dtype, copy=False)32 return output.numpy().astype(x_dtype, copy=False)
32 else:33 else:
33- return np.log1p(x)34+ return np.log1p(x)
35+ 
36+ 
37+def aclnn_log1p_golden(self, out=None, **kwargs):
38+ """
39+ Aclnn golden for aclnnLog1p.
40+ Parameters follow @aclnnLog1pGetWorkspaceSize without workspaceSize & executor.
41+ All the input Tensors are torch.Tensor.
42+ """
43+ import torch
44+ 
45+ return [torch.log1p(self)]
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,absolute_precision
2+Log1p_float32_ND_fuzz_1,aclnnLog1p,"('float32', 'float32')","('ND',)","((196, 2, 1, 76),(196, 2, 1, 76))","((-1000, -10),)","(-1,)","('float32',)",0.0001
3+Log1p_float16_ND_fuzz_2,aclnnLog1p,"('float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('float16',)",0.001
4+Log1p_float32_ND_fuzz_3,aclnnLog1p,"('float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+30, 3.4e+30),)","(-1,)","('float32',)",0.0001
5+Log1p_float16_ND_fuzz_4,aclnnLog1p,"('float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('float16',)",0.001
6+Log1p_float16_ND_fuzz_5,aclnnLog1p,"('float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)","('float16',)",0.001
@@ -11,14 +11,20 @@
11# ----------------------------------------------------------------------------11# ----------------------------------------------------------------------------
12 12 
13import numpy as np13import numpy as np
14+import torch
14 15 
15 16 
16__golden__ = {17__golden__ = {
17- "kernel": {18+ "aclnn": {
18- "logical_not": "logical_not_golden"19+ "aclnnLogicalNot": "aclnn_logical_not_golden",
19- }20+ },
21+ "kernel": {"logical_not": "logical_not_golden"},
20}22}
21 23 
22 24 
23def logical_not_golden(x, **kwargs):25def logical_not_golden(x, **kwargs):
24- return np.logical_not(x)26+ return np.logical_not(x)
27+ 
28+ 
29+def aclnn_logical_not_golden(self, out=None, **kwargs):
30+ return [torch.logical_not(self)]
@@ -0,0 +1,2 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,absolute_precision
2+LogicalNot_float32_ND_fuzz_1,aclnnLogicalNot,"('float32',)","('ND',)","((196, 2, 1, 76),(196, 2, 1, 76))","((-2, 2),)","(-1,)","('float32',)",0.0001
@@ -9,30 +9,42 @@
9# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.9# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
10# See LICENSE in the root of the software repository for the full text of the License.10# See LICENSE in the root of the software repository for the full text of the License.
11# ----------------------------------------------------------------------------11# ----------------------------------------------------------------------------
12-import numpy as np12+import torch
13 13 
14__golden__ = {14__golden__ = {
15- "kernel": {15+ "aclnn": {
16- "masked_scale": "masked_scale_golden"16+ "aclnnMaskedScale": "aclnn_masked_scale_golden",
17- }17+ },
18+ "kernel": {"masked_scale": "masked_scale_golden"},
18}19}
19- 20+ 
20-def masked_scale_golden(x, mask, value: float,21+ 
21- **kwargs):22+def masked_scale_golden(x, mask, value: float, **kwargs):
22- '''23+ """
23 Kernel golden for masked_scale.24 Kernel golden for masked_scale.
24 All the parameters follow @masked_scale_def.cpp without outputs.25 All the parameters follow @masked_scale_def.cpp without outputs.
25 All the input Tensors are numpy.ndarray.26 All the input Tensors are numpy.ndarray.
26- kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, 27+ kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes,
27- input_formats, output_formats, input_ori_formats, output_ori_formats,28+ input_formats, output_formats, input_ori_formats, output_ori_formats,
28- input_dtypes, output_dtypes.29+ input_dtypes, output_dtypes.
29- '''30+ """
30 x_dtype = x.dtype31 x_dtype = x.dtype
31 32 
32- if x_dtype.name not in ('float32', 'float64'):33+ if x_dtype.name not in ("float32", "float64"):
33- x = x.astype('float32')34+ x = x.astype("float32")
34- if mask.dtype.name not in ('float32', 'float64'):35+ if mask.dtype.name not in ("float32", "float64"):
35- mask = mask.astype('float32')36+ mask = mask.astype("float32")
36 res = x * mask * value37 res = x * mask * value
37 38 
38 return res.astype(x_dtype, copy=False)39 return res.astype(x_dtype, copy=False)
40+ 
41+ 
42+def aclnn_masked_scale_golden(self, mask, scale, out=None, **kwargs):
43+ if hasattr(scale, "item"):
44+ scale = scale.item()
45+ orig_dtype = self.dtype
46+ if orig_dtype in (torch.float16, torch.bfloat16):
47+ self = self.to(torch.float32)
48+ mask_f = mask.to(torch.float32)
49+ result = self * mask_f * scale
50+ return [result.to(orig_dtype)]
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,attributes,absolute_precision
2+MaskedScale_ND_fuzz_1,aclnnMaskedScale,"('float32', 'float32', 'float32')","('ND',)","((196, 2, 1, 76, 1, 1), (196, 2, 1, 76, 1, 1), (196, 2, 1, 76, 1, 1))","((-1000, -10),)","(-1,)",{'scale':0.12},0.0001
3+MaskedScale_ND_fuzz_2,aclnnMaskedScale,"('float16', 'float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)",{'scale':0.12},0.001
4+MaskedScale_ND_fuzz_3,aclnnMaskedScale,"('float32', 'float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)",{'scale':0.12},0.0001
5+MaskedScale_ND_fuzz_4,aclnnMaskedScale,"('float16', 'float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)",{'scale':0.12},0.001
6+MaskedScale_ND_fuzz_5,aclnnMaskedScale,"('float16', 'float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)",{'scale':0.12},0.001
@@ -15,19 +15,31 @@ import torch
15 15 
16 16 
17__golden__ = {17__golden__ = {
18- "kernel": {18+ "aclnn": {
19- "neg": "neg_golden"19+ "aclnnNeg": "aclnn_neg_golden",
20- }20+ },
21+ "kernel": {"neg": "neg_golden"},
21}22}
22 23 
23 24 
24def neg_golden(x, **kwargs):25def neg_golden(x, **kwargs):
25 ori_dtype = kwargs.get("input_dtypes", ["float32"])[0]26 ori_dtype = kwargs.get("input_dtypes", ["float32"])[0]
26 x_dtype = x.dtype27 x_dtype = x.dtype
27- 28+ 
28 if "bfloat16" in str(ori_dtype).lower() or "float16" in str(ori_dtype).lower():29 if "bfloat16" in str(ori_dtype).lower() or "float16" in str(ori_dtype).lower():
29 x_tensor = torch.from_numpy(x.astype(np.float32))30 x_tensor = torch.from_numpy(x.astype(np.float32))
30 output = torch.neg(x_tensor)31 output = torch.neg(x_tensor)
31 return output.numpy().astype(x_dtype, copy=False)32 return output.numpy().astype(x_dtype, copy=False)
32 else:33 else:
33- return np.negative(x)34+ return np.negative(x)
35+ 
36+ 
37+def aclnn_neg_golden(self, out=None, **kwargs):
38+ """
39+ Aclnn golden for aclnnNeg.
40+ Parameters follow @aclnnNegGetWorkspaceSize without workspaceSize & executor.
41+ All the input Tensors are torch.Tensor.
42+ """
43+ import torch
44+ 
45+ return [torch.neg(self)]
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,absolute_precision
2+Neg_float32_ND_fuzz_1,aclnnNeg,"('float32', 'float32')","('ND',)","((196, 2, 1, 76, 1, 1), (196, 2, 1, 76, 1, 1))","((-1000, -10),)","(-1,)",0.0001
3+Neg_float16_ND_fuzz_2,aclnnNeg,"('float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)",0.001
4+Neg_float32_ND_fuzz_3,aclnnNeg,"('float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)",0.0001
5+Neg_float16_ND_fuzz_4,aclnnNeg,"('float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)",0.001
6+Neg_float16_ND_fuzz_5,aclnnNeg,"('float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)",0.001
@@ -11,7 +11,14 @@
11# ----------------------------------------------------------------------------11# ----------------------------------------------------------------------------
12 12 
13 13 
14-__golden__ = {"kernel": {"reduce_max": "reduce_max_golden"}}14+__golden__ = {
15+ "aclnn": {
16+ "aclnnAmax": "aclnn_amax_golden",
17+ "aclnnMax": "aclnn_max_golden",
18+ "aclnnMaxV2": "aclnn_max_v2_golden",
19+ },
20+ "kernel": {"reduce_max": "reduce_max_golden"},
21+}
15 22 
16 23 
17def reduce_max_golden(x, axes=None, keep_dims: bool = False, **kwargs):24def reduce_max_golden(x, axes=None, keep_dims: bool = False, **kwargs):
@@ -55,3 +62,51 @@ def reduce_max_golden(x, axes=None, keep_dims: bool = False, **kwargs):
55 # Fallback to NumPy62 # Fallback to NumPy
56 res = np.max(x, axis=axis, keepdims=keep_dims)63 res = np.max(x, axis=axis, keepdims=keep_dims)
57 return res.astype(input_dtype, copy=False)64 return res.astype(input_dtype, copy=False)
65+ 
66+ 
67+def aclnn_max_v2_golden(
68+ self, dims=0, keepDims=0, noopWithEmptyDims=0, out=None, **kwargs
69+):
70+ """
71+ Aclnn golden for aclnnMaxV2.
72+ Parameters follow @aclnnMaxV2GetWorkspaceSize without workspaceSize & executor.
73+ All the input Tensors are torch.Tensor.
74+ """
75+ import torch
76+ 
77+ ipt = self
78+ dim = kwargs.get("attributes", {})["dims"]
79+ keepdim = kwargs.get("attributes", {})["keepDims"]
80+ noop_with_empty_dims = kwargs.get("attributes", {})["noopWithEmptyDims"]
81+ if dim is None or (isinstance(dim, (tuple, list)) and len(dim) == 0):
82+ if noop_with_empty_dims:
83+ result = ipt
84+ else:
85+ result = ipt.flatten()
86+ if keepdim:
87+ result = result.reshape([1] * ipt.dim())
88+ else:
89+ result = torch.amax(ipt, dim=dim, keepdim=keepdim)
90+ return result
91+ 
92+ 
93+def aclnn_max_golden(self, out=None, **kwargs):
94+ """
95+ Aclnn golden for aclnnMax.
96+ Parameters follow @aclnnMaxGetWorkspaceSize without workspaceSize & executor.
97+ All the input Tensors are torch.Tensor.
98+ """
99+ import torch
100+ 
101+ return [torch.ops.aten.amax(self)]
102+ 
103+ 
104+def aclnn_amax_golden(self, dim=0, keepDim=0, out=None, **kwargs):
105+ """
106+ Aclnn golden for aclnnAmax.
107+ Parameters follow @aclnnAmaxGetWorkspaceSize without workspaceSize & executor.
108+ All the input Tensors are torch.Tensor.
109+ """
110+ import torch
111+ 
112+ return torch.amax(self, dim=dim, keepdim=keepDim)
@@ -0,0 +1,2 @@
1+testcase_name,api_name,tensor_dtypes,attributes,tensor_view_shapes,output_tensor_indexes,input_data_ranges,precision_tolerances,absolute_precision
2+aclnnAmax_00,aclnnAmax,"('bfloat16', 'bfloat16')","{'dim': [-5, 4], 'keepDim': True}","((1,7831,1,1,107),(1,7831,1,1,1))","(1,)","[[0, 0.001]]","((0.001, 0.001),)",0.000001
@@ -0,0 +1,2 @@
1+testcase_name,api_name,tensor_dtypes,attributes,tensor_view_shapes,output_tensor_indexes,input_data_ranges,precision_tolerances,absolute_precision
2+aclnnMax_00,aclnnMax,"('float32', 'float32')",,"((2,3,4,32),(1,))","(1,)","[[0, 0.001]]","((0.001, 0.001),)",0.000001
@@ -11,47 +11,80 @@
11# ----------------------------------------------------------------------------11# ----------------------------------------------------------------------------
12 12 
13 13 
14-__golden__ = {"kernel": {"reduce_min": "reduce_min_golden"}}14+import torch
15 15 
16 16 
17-def reduce_min_golden(x, axes=None, keep_dims: bool = False, **kwargs):17+__golden__ = {
18+ "aclnn": {
19+ "aclnnAminmax": "aclnn_aminmax_golden",
20+ "aclnnAminmaxDim": "aclnn_aminmax_dim_golden",
21+ "aclnnAminmaxAll": "aclnn_aminmax_all_golden",
22+ },
23+ "kernel": {"reduce_min": "reduce_min_golden"},
24+}
25+ 
26+ 
27+# def reduce_min_golden(x, axes=None, keep_dims: bool = False, **kwargs):
28+# """
29+# Kernel golden for reduce_min.
30+# """
31+# import numpy as np
32+ 
33+# input_dtype = x.dtype
34+# if str(input_dtype) == "bfloat16":
35+# x = x.astype(np.float32)
36+ 
37+# if axes is not None:
38+# axis = tuple(int(a) for a in np.asarray(axes).flatten())
39+# else:
40+# axis = None
41+ 
42+# try:
43+# import tensorflow as tf
44+# x_tensor = tf.constant(x)
45+# res_tensor = tf.reduce_min(x_tensor, axis=axis, keepdims=keep_dims)
46+# res = res_tensor.numpy()
47+# except ImportError:
48+# try:
49+# import torch
50+# x_torch = torch.from_numpy(x)
51+# if axis is not None:
52+# res_torch = torch.amin(x_torch, dim=axis, keepdim=keep_dims)
53+# else:
54+# res_torch = torch.amin(x_torch, keepdim=keep_dims)
55+# res = res_torch.numpy()
56+# except ImportError:
57+# res = np.min(x, axis=axis, keepdims=keep_dims)
58+# return res.astype(input_dtype, copy=False)
59+ 
60+ 
61+def aclnn_aminmax_golden(self, dim=0, keepDim=0, minOut=None, maxOut=None, **kwargs):
18 """62 """
19- Kernel golden for reduce_min.63+ Aclnn golden for aclnnAminmax.
20- All the parameters follow @reduce_min_def.cpp without outputs.
21- All the input Tensors are numpy.ndarray.
22- kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes,
23- input_formats, output_formats, input_ori_formats, output_ori_formats,
24- input_dtypes, output_dtypes.
25 """64 """
26- import numpy as np65+ if isinstance(dim, (tuple, list)):
27- 66+ min_val = torch.amin(self, dim=dim, keepdim=bool(keepDim))
28- input_dtype = x.dtype67+ max_val = torch.amax(self, dim=dim, keepdim=bool(keepDim))
29- if str(input_dtype) == "bfloat16":
30- x = x.astype(np.float32)
31- 
32- if axes is not None:
33- axis = tuple(int(a) for a in np.asarray(axes).flatten())
34 else:68 else:
35- axis = None69+ result = torch.aminmax(self, dim=dim[0], keepdim=bool(keepDim))
70+ min_val = result.min
71+ max_val = result.max
72+ return [min_val, max_val]
36 73 
37- # Try TensorFlow first (supports empty tensors), fallback to PyTorch, then NumPy
38- try:
39- import tensorflow as tf
40 74 
41- x_tensor = tf.constant(x)75+def aclnn_aminmax_all_golden(self, minOut=None, maxOut=None, **kwargs):
42- res_tensor = tf.reduce_min(x_tensor, axis=axis, keepdims=keep_dims)76+ """
43- res = res_tensor.numpy()77+ Aclnn golden for aclnnAminmaxAll.
44- except ImportError:78+ """
45- try:79+ result = torch.aminmax(self)
46- import torch80+ return [result.min, result.max]
47 81 
48- x_torch = torch.from_numpy(x)82+ 
49- if axis is not None:83+def aclnn_aminmax_dim_golden(
50- res_torch = torch.amin(x_torch, dim=axis, keepdim=keep_dims)84+ self, dim=0, keepDim=0, minOut=None, maxOut=None, **kwargs
51- else:85+):
52- res_torch = torch.amin(x_torch, keepdim=keep_dims)86+ """
53- res = res_torch.numpy()87+ Aclnn golden for aclnnAminmaxDim.
54- except ImportError:88+ """
55- # Fallback to NumPy89+ result = torch.aminmax(self, dim=dim, keepdim=bool(keepDim))
56- res = np.min(x, axis=axis, keepdims=keep_dims)90+ return [result.min, result.max]
57- return res.astype(input_dtype, copy=False)
@@ -0,0 +1,2 @@
1+testcase_name,api_name,tensor_dtypes,attributes,tensor_view_shapes,output_tensor_indexes,input_data_ranges,precision_tolerances,absolute_precision
2+aclnnAminmaxAll_00,aclnnAminmaxAll,"('float32', 'float32', 'float32')",,"((2,35),(1,),(1,))","(1,2)","[[0, 0.001]]","((0.001, 0.001),)",0.000001
@@ -0,0 +1,2 @@
1+testcase_name,api_name,tensor_dtypes,attributes,tensor_view_shapes,output_tensor_indexes,input_data_ranges,precision_tolerances,absolute_precision
2+aclnnAminmaxDim_00,aclnnAminmaxDim,"('float32', 'float32', 'float32')","{'dim': 4, 'keepDim': True}","((1,7831,1,1,107),(1,7831,1,1,1),(1,7831,1,1,1))","(1,2)","[[0, 0.001]]","((0.001, 0.001),)",0.000001
@@ -0,0 +1,2 @@
1+testcase_name,api_name,tensor_dtypes,attributes,tensor_view_shapes,output_tensor_indexes,input_data_ranges,precision_tolerances,absolute_precision
2+aclnnAminmax_00,aclnnAminmax,"('float32', 'float32', 'float32')","{'dim': [4,], 'keepDim': True}","((1,7831,1,1,107),(1,7831,1,1,1),(1,7831,1,1,1))","(1,2)","[[0, 0.001]]","((0.001, 0.001),)",0.000001
@@ -4,7 +4,7 @@
4# Copyright (c) 2026 Huawei Technologies Co., Ltd.4# Copyright (c) 2026 Huawei Technologies Co., Ltd.
5# This program is free software, you can redistribute it and/or modify it under the terms and conditions of5# This program is free software, you can redistribute it and/or modify it under the terms and conditions of
6# CANN Open Software License Agreement Version 2.0 (the "License").6# CANN Open Software License Agreement Version 2.0 (the "License").
7-# Please refer to the License for details. You may not use your file except compliance with the License.7+# Please refer to the License for details. You may not use this file except in compliance with the License.
8# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,8# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
9# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.9# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
10# See LICENSE in the root of the software repository for the full text of the License.10# See LICENSE in the root of the software repository for the full text of the License.
@@ -14,26 +14,28 @@ import numpy as np
14 14 
15 15 
16__golden__ = {16__golden__ = {
17- "kernel": {17+ "aclnn": {
18- "rsqrt": "rsqrt_golden"18+ "aclnnInplaceRsqrt": "aclnn_inplace_rsqrt_golden",
19- }19+ "aclnnRsqrt": "aclnn_rsqrt_golden",
20+ },
21+ "kernel": {"rsqrt": "rsqrt_golden"},
20}22}
21 23 
22 24 
23def rsqrt_golden(x, **kwargs):25def rsqrt_golden(x, **kwargs):
24- '''26+ """
25 Kernel golden for rsqrt.27 Kernel golden for rsqrt.
26 All the parameters follow @rsqrt_def.cpp without outputs.28 All the parameters follow @rsqrt_def.cpp without outputs.
27 All the input Tensors are numpy.ndarray.29 All the input Tensors are numpy.ndarray.
28 kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes,30 kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes,
29 input_formats, output_formats, input_ori_formats, output_ori_formats,31 input_formats, output_formats, input_ori_formats, output_ori_formats,
30 input_dtypes, output_dtypes.32 input_dtypes, output_dtypes.
31- '''33+ """
32 import torch34 import torch
33- 35+ 
34 ori_dtype = kwargs.get("input_dtypes", ["float32"])[0]36 ori_dtype = kwargs.get("input_dtypes", ["float32"])[0]
35 x_dtype = x.dtype37 x_dtype = x.dtype
36- 38+ 
37 if ori_dtype and "bfloat16" in str(ori_dtype).lower():39 if ori_dtype and "bfloat16" in str(ori_dtype).lower():
38 x_tensor = torch.from_numpy(x.astype(np.float32))40 x_tensor = torch.from_numpy(x.astype(np.float32))
39 output = torch.rsqrt(x_tensor)41 output = torch.rsqrt(x_tensor)
@@ -45,4 +47,44 @@ def rsqrt_golden(x, **kwargs):
45 else:47 else:
46 x_tensor = torch.from_numpy(x)48 x_tensor = torch.from_numpy(x)
47 output = torch.rsqrt(x_tensor)49 output = torch.rsqrt(x_tensor)
48- return output.numpy()50+ return output.numpy()
51+ 
52+ 
53+def aclnn_rsqrt_golden(self, out=None, **kwargs):
54+ """
55+ Aclnn golden for aclnnRsqrt.
56+ Parameters follow @aclnnRsqrtGetWorkspaceSize without workspaceSize & executor.
57+ All the input Tensors are torch.Tensor.
58+ """
59+ import torch
60+ 
61+ x = self
62+ x_dtype = x.dtype
63+ if x_dtype == torch.float16 or x_dtype == torch.bfloat16:
64+ y = torch.ops.aten.rsqrt(x.to(torch.float32))
65+ else:
66+ y = torch.ops.aten.rsqrt(x)
67+ 
68+ if x_dtype == torch.float16 or x_dtype == torch.bfloat16:
69+ y = y.to(x_dtype)
70+ return y
71+ 
72+ 
73+def aclnn_inplace_rsqrt_golden(selfRef=None, **kwargs):
74+ """
75+ Aclnn golden for aclnnInplaceRsqrt.
76+ Parameters follow @aclnnInplaceRsqrtGetWorkspaceSize without workspaceSize & executor.
77+ All the input Tensors are torch.Tensor.
78+ """
79+ import torch
80+ 
81+ x = selfRef
82+ x_dtype = x.dtype
83+ if x_dtype == torch.float16 or x_dtype == torch.bfloat16:
84+ y = torch.ops.aten.rsqrt(x.to(torch.float32))
85+ else:
86+ y = torch.ops.aten.rsqrt(x)
87+ 
88+ if x_dtype == torch.float16 or x_dtype == torch.bfloat16:
89+ y = y.to(x_dtype)
90+ return y
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,absolute_precision
2+InplaceRsqrt_float32_ND_fuzz_1,aclnnInplaceRsqrt,('float32'),"('ND',)","((196, 2, 1, 76, 1, 1))","((0.001,1000),)","(-1,)","('float32',)",0.0001
3+InplaceRsqrt_float16_ND_fuzz_2,aclnnInplaceRsqrt,('float16'),"('ND',)","((192, 64, 1, 1, 1, 5, 1))","((0.001,1000),)","(-1,)","('float16',)",0.001
4+InplaceRsqrt_float32_ND_fuzz_3,aclnnInplaceRsqrt,('float32'),"('ND',)","((1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)","('float32',)",0.0001
5+InplaceRsqrt_float16_ND_fuzz_4,aclnnInplaceRsqrt,('float16'),"('ND',)","((139, 3, 1, 1, 1, 192))","((0.001,1000),)","(-1,)","('float16',)",0.001
6+InplaceRsqrt_float16_ND_fuzz_5,aclnnInplaceRsqrt,('float16'),"('ND',)","((1, 139, 1, 28, 1, 22, 1))","((0.001,1000),)","(-1,)","('float16',)",0.001
@@ -0,0 +1,6 @@
1+,testcase_name,api_name,tensor_view_shapes,tensor_dtypes
2+0,aclnnRsqrt_00000,aclnnRsqrt,"[[46, 40, 9, 3, 24, 36],[46, 40, 9, 3, 24, 36]]","('float32','float32')"
3+1,aclnnRsqrt_00001,aclnnRsqrt,"[[34, 22, 39, 36],[34, 22, 39, 36]]","('bfloat16','bfloat16')"
4+2,aclnnRsqrt_00002,aclnnRsqrt,"[[3],[3]]","('float32','float32')"
5+3,aclnnRsqrt_00003,aclnnRsqrt,"[[17, 30, 49, 11],[17, 30, 49, 11]]","('float32','float32')"
6+4,aclnnRsqrt_00004,aclnnRsqrt,"[[16, 25],[16, 25]]","('float16','float16')"
@@ -14,26 +14,27 @@ import numpy as np
14 14 
15 15 
16__golden__ = {16__golden__ = {
17- "kernel": {17+ "aclnn": {
18- "sign": "sign_golden"18+ "aclnnSign": "aclnn_sign_golden",
19- }19+ },
20+ "kernel": {"sign": "sign_golden"},
20}21}
21 22 
22 23 
23def sign_golden(x, **kwargs):24def sign_golden(x, **kwargs):
24- '''25+ """
25 Kernel golden for sign.26 Kernel golden for sign.
26 All the parameters follow @sign_def.cpp without outputs.27 All the parameters follow @sign_def.cpp without outputs.
27 All the input Tensors are numpy.ndarray.28 All the input Tensors are numpy.ndarray.
28 kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes,29 kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes,
29 input_formats, output_formats, input_ori_formats, output_ori_formats,30 input_formats, output_formats, input_ori_formats, output_ori_formats,
30 input_dtypes, output_dtypes.31 input_dtypes, output_dtypes.
31- '''32+ """
32 import torch33 import torch
33- 34+ 
34 ori_dtype = kwargs.get("input_dtypes", ["float32"])[0]35 ori_dtype = kwargs.get("input_dtypes", ["float32"])[0]
35 x_dtype = x.dtype36 x_dtype = x.dtype
36- 37+ 
37 if ori_dtype and "bfloat16" in str(ori_dtype).lower():38 if ori_dtype and "bfloat16" in str(ori_dtype).lower():
38 x_tensor = torch.from_numpy(x.astype(np.float32))39 x_tensor = torch.from_numpy(x.astype(np.float32))
39 output = torch.sign(x_tensor)40 output = torch.sign(x_tensor)
@@ -45,4 +46,15 @@ def sign_golden(x, **kwargs):
45 else:46 else:
46 x_tensor = torch.from_numpy(x)47 x_tensor = torch.from_numpy(x)
47 output = torch.sign(x_tensor)48 output = torch.sign(x_tensor)
48- return output.numpy()49+ return output.numpy()
50+ 
51+ 
52+def aclnn_sign_golden(self, result=None, **kwargs):
53+ """
54+ Aclnn golden for aclnnSign.
55+ Parameters follow @aclnnSignGetWorkspaceSize without workspaceSize & executor.
56+ All the input Tensors are torch.Tensor.
57+ """
58+ import torch
59+ 
60+ return [torch.sign(self)]
@@ -0,0 +1,6 @@
1+,testcase_name,api_name,tensor_view_shapes,tensor_dtypes
2+0,aclnnSign_00000,aclnnSign,"[[23],[23]]","('float16','float16')"
3+1,aclnnSign_00001,aclnnSign,"[[3, 11, 7],[3, 11, 7]]","('int32','int32')"
4+2,aclnnSign_00002,aclnnSign,"[[26, 1, 42, 25, 45],[26, 1, 42, 25, 45]]","('int64','int64')"
5+3,aclnnSign_00003,aclnnSign,"[[9, 17, 44],[9, 17, 44]]","('bfloat16','bfloat16')"
6+4,aclnnSign_00004,aclnnSign,"[[22, 45, 19],[22, 45, 19]]","('float32','float32')"
@@ -11,29 +11,31 @@
11# ----------------------------------------------------------------------------11# ----------------------------------------------------------------------------
12 12 
13import numpy as np13import numpy as np
14+import torch
14 15 
15 16 
16__golden__ = {17__golden__ = {
17- "kernel": {18+ "aclnn": {
18- "sin": "sin_golden"19+ "aclnnSin": "aclnn_sin_golden",
19- }20+ },
21+ "kernel": {"sin": "sin_golden"},
20}22}
21 23 
22 24 
23def sin_golden(x, **kwargs):25def sin_golden(x, **kwargs):
24- '''26+ """
25 Kernel golden for sin.27 Kernel golden for sin.
26 All the parameters follow @sin_def.cpp without outputs.28 All the parameters follow @sin_def.cpp without outputs.
27 All the input Tensors are numpy.ndarray.29 All the input Tensors are numpy.ndarray.
28 kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes,30 kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes,
29 input_formats, output_formats, input_ori_formats, output_ori_formats,31 input_formats, output_formats, input_ori_formats, output_ori_formats,
30 input_dtypes, output_dtypes.32 input_dtypes, output_dtypes.
31- '''33+ """
32 import torch34 import torch
33- 35+ 
34 ori_dtype = kwargs.get("input_dtypes", ["float32"])[0]36 ori_dtype = kwargs.get("input_dtypes", ["float32"])[0]
35 x_dtype = x.dtype37 x_dtype = x.dtype
36- 38+ 
37 if ori_dtype and "bfloat16" in str(ori_dtype).lower():39 if ori_dtype and "bfloat16" in str(ori_dtype).lower():
38 x_tensor = torch.from_numpy(x.astype(np.float32))40 x_tensor = torch.from_numpy(x.astype(np.float32))
39 output = torch.sin(x_tensor)41 output = torch.sin(x_tensor)
@@ -43,4 +45,8 @@ def sin_golden(x, **kwargs):
43 output = torch.sin(x_tensor)45 output = torch.sin(x_tensor)
44 return output.numpy().astype(x_dtype, copy=False)46 return output.numpy().astype(x_dtype, copy=False)
45 else:47 else:
46- return np.sin(x)48+ return np.sin(x)
49+ 
50+ 
51+def aclnn_sin_golden(input, out=None, **kwargs):
52+ return [torch.sin(input)]
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,absolute_precision
2+Sin_float32_ND_fuzz_1,aclnnSin,"('float32', 'float32')","('ND',)","((196, 2, 1, 76),(196, 2, 1, 76))","((-1000, -10),)","(-1,)","('float32',)",0.0001
3+Sin_float16_ND_fuzz_2,aclnnSin,"('float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('float16',)",0.001
4+Sin_float32_ND_fuzz_3,aclnnSin,"('float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)","('float32',)",0.0001
5+Sin_float16_ND_fuzz_4,aclnnSin,"('float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('float16',)",0.001
6+Sin_float16_ND_fuzz_5,aclnnSin,"('float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)","('float16',)",0.001
@@ -0,0 +1,27 @@
1+#!/usr/bin/env python3
2+# -*- coding: UTF-8 -*-
3+# ----------------------------------------------------------------------------
4+# Copyright (c) 2026 Huawei Technologies Co., Ltd.
5+# This program is free software, you can redistribute it and/or modify it under the terms and conditions of
6+# CANN Open Software License Agreement Version 2.0 (the "License").
7+# Please refer to the License for details. You may not use this file except in compliance with the License.
8+# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
9+# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
10+# See LICENSE in the root of the software repository for the full text of the License.
11+# ----------------------------------------------------------------------------
12+import torch
13+ 
14+__golden__ = {
15+ "aclnn": {
16+ "aclnnSinh": "aclnn_sinh_golden",
17+ }
18+}
19+ 
20+ 
21+def aclnn_sinh_golden(self, out=None, **kwargs):
22+ """
23+ Aclnn golden for aclnnSinh.
24+ Parameters follow @aclnnSinhGetWorkspaceSize without workspaceSize & executor.
25+ All the input Tensors are torch.Tensor.
26+ """
27+ return [torch.sinh(self)]
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_view_shapes,tensor_dtypes,input_data_ranges,precision_tolerances
2+aclnnSinh_fp32_1d_000001_0001,aclnnSinh,"((1,),(1,))","('float32','float32')","((-0.5,0.5),(-0.5,0.5))","(0.0001,0.0001)"
3+aclnnSinh_fp32_1d_000002_0002,aclnnSinh,"((2,),(2,))","('float32','float32')","((-0.5,0.5),(-0.5,0.5))","(0.0001,0.0001)"
4+aclnnSinh_fp32_1d_000003_0003,aclnnSinh,"((3,),(3,))","('float32','float32')","((-3.14,3.14),(-3.14,3.14))","(0.0001,0.0001)"
5+aclnnSinh_fp32_1d_000004_0004,aclnnSinh,"((4,),(4,))","('float32','float32')","((-1,1),(-1,1))","(0.0001,0.0001)"
6+aclnnSinh_fp32_1d_000005_0005,aclnnSinh,"((5,),(5,))","('float32','float32')","((-1,1),(-1,1))","(0.0001,0.0001)"
@@ -0,0 +1,35 @@
1+#!/usr/bin/env python3
2+# -*- coding: UTF-8 -*-
3+# ----------------------------------------------------------------------------
4+# Copyright (c) 2026 Huawei Technologies Co., Ltd.
5+# This program is free software, you can redistribute it and/or modify it under the terms and conditions of
6+# CANN Open Software License Agreement Version 2.0 (the "License").
7+# Please refer to the License for details. You may not use this file except in compliance with the License.
8+# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
9+# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
10+# See LICENSE in the root of the software repository for the full text of the License.
11+# ----------------------------------------------------------------------------
12+import torch
13+ 
14+__golden__ = {
15+ "aclnn": {
16+ "aclnnSort": "aclnn_sort_golden",
17+ }
18+}
19+ 
20+ 
21+def aclnn_sort_golden(
22+ self, stable=0, dim=0, descending=0, valuesOut=None, indicesOut=None, **kwargs
23+):
24+ """
25+ Aclnn golden for aclnnSort.
26+ Parameters follow @aclnnSortGetWorkspaceSize without workspaceSize & executor.
27+ All the input Tensors are torch.Tensor.
28+ """
29+ input_x = self
30+ 
31+ stable = stable
32+ dim = dim
33+ descending = descending
34+ y1, y2 = torch.sort(input=input_x, dim=dim, descending=descending, stable=True)
35+ return y1, y2
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,attributes,tensor_view_shapes,output_tensor_indexes,input_data_ranges,is_enabled
2+aclnn_sort_001,aclnnSort,"('float16','float16','int64')","('ND', 'ND', 'ND')","{ 'stable': True,'dim': -1, 'descending': True}","((1000, 14840), (1000, 14840), (1000,14840))","(1,2)","((-200,200),)",1
3+aclnn_sort_002,aclnnSort,"('bfloat16','bfloat16','int64')","('ND', 'ND', 'ND')","{ 'stable': True,'dim': -1, 'descending': False}","((1000, 14840), (1000, 14840), (1000,14840))","(1,2)","((-200,200),)",1
4+aclnn_sort_003,aclnnSort,"('float32','float32','int64')","('ND', 'ND', 'ND')","{ 'stable': True,'dim': -1, 'descending': False}","((3, 32768), (3, 32768), (3,32768))","(1,2)","((-200,200),)",1
5+aclnn_sort_004,aclnnSort,"('int32','int32','int64')","('ND', 'ND', 'ND')","{ 'stable': True,'dim': -1, 'descending': True}","((3, 32768), (3, 32768), (3,32768))","(1,2)","((-200,200),)",1
6+aclnn_sort_005,aclnnSort,"('int16','int16','int64')","('ND', 'ND', 'ND')","{ 'stable': True,'dim': -1, 'descending': False}","((3, 32768), (3, 32768), (3,32768))","(1,2)","((-200,200),)",1
@@ -14,26 +14,27 @@ import numpy as np
14 14 
15 15 
16__golden__ = {16__golden__ = {
17- "kernel": {17+ "aclnn": {
18- "sqrt": "sqrt_golden"18+ "aclnnSqrt": "aclnn_sqrt_golden",
19- }19+ },
20+ "kernel": {"sqrt": "sqrt_golden"},
20}21}
21 22 
22 23 
23def sqrt_golden(x, **kwargs):24def sqrt_golden(x, **kwargs):
24- '''25+ """
25 Kernel golden for sqrt.26 Kernel golden for sqrt.
26 All the parameters follow @sqrt_def.cpp without outputs.27 All the parameters follow @sqrt_def.cpp without outputs.
27 All the input Tensors are numpy.ndarray.28 All the input Tensors are numpy.ndarray.
28 kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes,29 kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes,
29 input_formats, output_formats, input_ori_formats, output_ori_formats,30 input_formats, output_formats, input_ori_formats, output_ori_formats,
30 input_dtypes, output_dtypes.31 input_dtypes, output_dtypes.
31- '''32+ """
32 import torch33 import torch
33- 34+ 
34 ori_dtype = kwargs.get("input_dtypes", ["float32"])[0]35 ori_dtype = kwargs.get("input_dtypes", ["float32"])[0]
35 x_dtype = x.dtype36 x_dtype = x.dtype
36- 37+ 
37 if ori_dtype and "bfloat16" in str(ori_dtype).lower():38 if ori_dtype and "bfloat16" in str(ori_dtype).lower():
38 x_tensor = torch.from_numpy(x.astype(np.float32))39 x_tensor = torch.from_numpy(x.astype(np.float32))
39 result = torch.sqrt(x_tensor)40 result = torch.sqrt(x_tensor)
@@ -45,4 +46,15 @@ def sqrt_golden(x, **kwargs):
45 else:46 else:
46 x_tensor = torch.from_numpy(x)47 x_tensor = torch.from_numpy(x)
47 result = torch.sqrt(x_tensor)48 result = torch.sqrt(x_tensor)
48- return result.numpy()49+ return result.numpy()
50+ 
51+ 
52+def aclnn_sqrt_golden(self, out=None, opExecutor=0, **kwargs):
53+ """
54+ Aclnn golden for aclnnSqrt.
55+ Parameters follow @aclnnSqrtGetWorkspaceSize without workspaceSize & executor.
56+ All the input Tensors are torch.Tensor.
57+ """
58+ import torch
59+ 
60+ return [torch.sqrt(self)]
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,absolute_precision
2+Sqrt_float32_ND_fuzz_1,aclnnSqrt,"('float32', 'float32')","('ND',)","((196, 2, 1, 76, 1, 1), (196, 2, 1, 76, 1, 1))","((-1000, -10),)","(-1,)",0.0001
3+Sqrt_float16_ND_fuzz_2,aclnnSqrt,"('float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)",0.001
4+Sqrt_float32_ND_fuzz_3,aclnnSqrt,"('float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)",0.0001
5+Sqrt_float16_ND_fuzz_4,aclnnSqrt,"('float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)",0.001
6+Sqrt_float16_ND_fuzz_5,aclnnSqrt,"('float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)",0.001
@@ -15,19 +15,31 @@ import torch
15 15 
16 16 
17__golden__ = {17__golden__ = {
18- "kernel": {18+ "aclnn": {
19- "square": "square_golden"19+ "aclnnSquare": "aclnn_square_golden",
20- }20+ },
21+ "kernel": {"square": "square_golden"},
21}22}
22 23 
23 24 
24def square_golden(x, **kwargs):25def square_golden(x, **kwargs):
25 ori_dtype = kwargs.get("input_dtypes", ["float32"])[0]26 ori_dtype = kwargs.get("input_dtypes", ["float32"])[0]
26 x_dtype = x.dtype27 x_dtype = x.dtype
27- 28+ 
28 if "bfloat16" in str(ori_dtype).lower() or "float16" in str(ori_dtype).lower():29 if "bfloat16" in str(ori_dtype).lower() or "float16" in str(ori_dtype).lower():
29 x_tensor = torch.from_numpy(x.astype(np.float32))30 x_tensor = torch.from_numpy(x.astype(np.float32))
30 output = torch.square(x_tensor)31 output = torch.square(x_tensor)
31 return output.numpy().astype(x_dtype, copy=False)32 return output.numpy().astype(x_dtype, copy=False)
32 else:33 else:
33- return np.square(x)34+ return np.square(x)
35+ 
36+ 
37+def aclnn_square_golden(self, out=None, **kwargs):
38+ """
39+ Aclnn golden for aclnnSquare.
40+ Parameters follow @aclnnSquareGetWorkspaceSize without workspaceSize & executor.
41+ All the input Tensors are torch.Tensor.
42+ """
43+ import torch
44+ 
45+ return [torch.square(self)]
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,absolute_precision
2+Square_float32_ND_fuzz_1,aclnnSquare,"('float32', 'float32')","('ND',)","((196, 2, 1, 76),(196, 2, 1, 76))","((-1000, -10),)","(-1,)","('float32',)",0.0001
3+Square_float16_ND_fuzz_2,aclnnSquare,"('float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('float16',)",0.001
4+Square_float32_ND_fuzz_3,aclnnSquare,"('float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)","('float32',)",0.0001
5+Square_float16_ND_fuzz_4,aclnnSquare,"('float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('float16',)",0.001
6+Square_float16_ND_fuzz_5,aclnnSquare,"('float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)","('float16',)",0.001
@@ -0,0 +1,31 @@
1+#!/usr/bin/env python3
2+# -*- coding: UTF-8 -*-
3+# ----------------------------------------------------------------------------
4+# Copyright (c) 2026 Huawei Technologies Co., Ltd.
5+# This program is free software, you can redistribute it and/or modify it under the terms and conditions of
6+# CANN Open Software License Agreement Version 2.0 (the "License").
7+# Please refer to the License for details. You may not use this file except in compliance with the License.
8+# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
9+# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
10+# See LICENSE in the root of the software repository for the full text of the License.
11+# ----------------------------------------------------------------------------
12+import torch
13+ 
14+__golden__ = {
15+ "aclnn": {
16+ "aclnnTanh": "aclnn_tanh_golden",
17+ }
18+}
19+ 
20+ 
21+def aclnn_tanh_golden(self, out=None, **kwargs):
22+ """
23+ Aclnn golden for aclnnTanh.
24+ Parameters follow @aclnnTanhGetWorkspaceSize without workspaceSize & executor.
25+ All the input Tensors are torch.Tensor.
26+ """
27+ orig_dtype = self.dtype
28+ if orig_dtype in (torch.float16, torch.bfloat16):
29+ self = self.to(torch.float32)
30+ result = torch.tanh(self)
31+ return [result.to(orig_dtype)]
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,absolute_precision
2+Tanh_float32_ND_fuzz_1,aclnnTanh,"('float32', 'float32')","('ND',)","((196, 2, 1, 76),(196, 2, 1, 76))","((-1000, -10),)","(-1,)","('float32',)",0.0001
3+Tanh_float16_ND_fuzz_2,aclnnTanh,"('float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('float16',)",0.001
4+Tanh_float32_ND_fuzz_3,aclnnTanh,"('float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)","('float32',)",0.0001
5+Tanh_float16_ND_fuzz_4,aclnnTanh,"('float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('float16',)",0.001
6+Tanh_float16_ND_fuzz_5,aclnnTanh,"('float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)","('float16',)",0.001
@@ -11,35 +11,41 @@
11# ----------------------------------------------------------------------------11# ----------------------------------------------------------------------------
12 12 
13import numpy as np13import numpy as np
14+import torch
14 15 
15 16 
16__golden__ = {17__golden__ = {
17- "kernel": {18+ "aclnn": {
18- "tensor_equal": "tensor_equal_golden"19+ "aclnnEqual": "aclnn_equal_golden",
19- }20+ },
21+ "kernel": {"tensor_equal": "tensor_equal_golden"},
20}22}
21 23 
22 24 
23def tensor_equal_golden(input_x, input_y, **kwargs):25def tensor_equal_golden(input_x, input_y, **kwargs):
24- '''26+ """
25 Kernel golden for tensor_equal.27 Kernel golden for tensor_equal.
26 All the parameters follow @tensor_equal_def.cpp without outputs.28 All the parameters follow @tensor_equal_def.cpp without outputs.
27 All the input Tensors are numpy.ndarray.29 All the input Tensors are numpy.ndarray.
28 kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes,30 kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes,
29 input_formats, output_formats, input_ori_formats, output_ori_formats,31 input_formats, output_formats, input_ori_formats, output_ori_formats,
30 input_dtypes, output_dtypes.32 input_dtypes, output_dtypes.
31- '''33+ """
32 import torch34 import torch
33- 35+ 
34 x_dtype = input_x.dtype36 x_dtype = input_x.dtype
35 y_dtype = input_y.dtype37 y_dtype = input_y.dtype
36- 38+ 
37 if str(x_dtype) == "bfloat16" and str(y_dtype) == "bfloat16":39 if str(x_dtype) == "bfloat16" and str(y_dtype) == "bfloat16":
38 input_x = input_x.astype(np.float32)40 input_x = input_x.astype(np.float32)
39 input_y = input_y.astype(np.float32)41 input_y = input_y.astype(np.float32)
40- 42+ 
41 tensor_x = torch.tensor(input_x)43 tensor_x = torch.tensor(input_x)
42 tensor_y = torch.tensor(input_y)44 tensor_y = torch.tensor(input_y)
43 result = torch.equal(tensor_x, tensor_y)45 result = torch.equal(tensor_x, tensor_y)
44- 46+ 
45- return np.array(result)47+ return np.array(result)
48+ 
49+ 
50+def aclnn_equal_golden(self, other, out=None, **kwargs):
51+ return [torch.tensor(torch.equal(self, other))]
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_view_shapes,tensor_formats,tensor_dtypes,input_data_ranges,is_enabled
2+aclnnEqual_test000001,aclnnEqual,"((229,3751),(229,3751),(1,),)","('ND',)","('float16', 'float16', 'bool' )","((2, 10), (-1, 0))",TRUE
3+aclnnEqual_test000002,aclnnEqual,"((5743,746,6),(5743,746,6),(1,),)","('ND',)","('int8', 'int8', 'bool' )","((-2, -1), (-10, -2))",TRUE
4+aclnnEqual_test000003,aclnnEqual,"((3,52,5,3591),(3,52,5,3591),(1,),)","('ND',)","('int16', 'int16', 'bool' )","((-1, 1), (-100, 100))",TRUE
5+aclnnEqual_test000004,aclnnEqual,"((12,7,14,7,21),(12,7,14,7,21),(1,),)","('ND',)","('uint8', 'uint8', 'bool' )","((-10, 10), (0, 0))",TRUE
6+aclnnEqual_test000005,aclnnEqual,"((6,4,15,1,7,18),(6,4,15,1,7,18),(1,),)","('ND',)","('uint16', 'uint16', 'bool' )","((0, 127), (127, 127))",TRUE
@@ -14,24 +14,32 @@ import numpy as np
14 14 
15 15 
16__golden__ = {16__golden__ = {
17- "kernel": {17+ "aclnn": {
18- "tile": "tile_golden",18+ "aclnnRepeat": "aclnn_repeat_golden",
19- "tile_d": "tile_golden"19+ },
20- }20+ "kernel": {"tile": "tile_golden", "tile_d": "tile_golden"},
21}21}
22 22 
23 23 
24def tile_golden(x, multiples, **kwargs):24def tile_golden(x, multiples, **kwargs):
25- '''25+ """
26 Kernel golden for tile / tile_d.26 Kernel golden for tile / tile_d.
27 All the parameters follow @tile_def.cpp without outputs.27 All the parameters follow @tile_def.cpp without outputs.
28 All the input Tensors are numpy.ndarray.28 All the input Tensors are numpy.ndarray.
29 kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes,29 kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes,
30 input_formats, output_formats, input_ori_formats, output_ori_formats,30 input_formats, output_formats, input_ori_formats, output_ori_formats,
31 input_dtypes, output_dtypes.31 input_dtypes, output_dtypes.
32- '''32+ """
33 multiples_arr = np.array(multiples).astype(np.int64)33 multiples_arr = np.array(multiples).astype(np.int64)
34 multiples_val = multiples_arr.tolist()34 multiples_val = multiples_arr.tolist()
35 if isinstance(multiples_val, int):35 if isinstance(multiples_val, int):
36 multiples_val = (multiples_val,)36 multiples_val = (multiples_val,)
37 return np.tile(x, multiples_val)37 return np.tile(x, multiples_val)
38+ 
39+ 
40+def aclnn_repeat_golden(self, repeats=0, out=None, **kwargs):
41+ if hasattr(repeats, "tolist"):
42+ repeats = repeats.tolist()
43+ elif isinstance(repeats, int):
44+ repeats = [repeats]
45+ return [self.repeat(*repeats)]
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_view_shapes,tensor_dtypes,attributes,output_tensor_indexes,precision_tolerances
2+tile_001,aclnnRepeat,"((2, 128, 1, 1),(2, 128, 28, 28))","('bfloat16','bfloat16')","{'repeats': [1, 1, 28, 28]}","(1,)",
3+tile_002,aclnnRepeat,"((2, 128, 1, 1),(2, 128, 28, 28))","('bool','bool')","{'repeats': [1, 1, 28, 28]}","(1,)",
4+tile_003,aclnnRepeat,"((1, 1, 32, 1, 1),(1, 43, 32, 1, 1))","('float16','float16')","{'repeats': [1, 43, 1, 1, 1]}","(1,)",
5+tile_004,aclnnRepeat,"((2, 128, 28, 1, 1),(2, 128, 28, 28, 1))","('float32','float32')","{'repeats': [1, 1, 1, 28, 1]}","(1,)",
6+tile_005,aclnnRepeat,"((2, 128, 1, 1),(2, 128, 28, 28))","('int32','int32')","{'repeats': [1, 1, 28, 28]}","(1,)",
@@ -14,26 +14,27 @@ import numpy as np
14 14 
15 15 
16__golden__ = {16__golden__ = {
17- "kernel": {17+ "aclnn": {
18- "trunc": "trunc_golden"18+ "aclnnTrunc": "aclnn_trunc_golden",
19- }19+ },
20+ "kernel": {"trunc": "trunc_golden"},
20}21}
21 22 
22 23 
23def trunc_golden(input_x, **kwargs):24def trunc_golden(input_x, **kwargs):
24- '''25+ """
25 Kernel golden for trunc.26 Kernel golden for trunc.
26 All the parameters follow @trunc_def.cpp without outputs.27 All the parameters follow @trunc_def.cpp without outputs.
27 All the input Tensors are numpy.ndarray.28 All the input Tensors are numpy.ndarray.
28 kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes,29 kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes,
29 input_formats, output_formats, input_ori_formats, output_ori_formats,30 input_formats, output_formats, input_ori_formats, output_ori_formats,
30 input_dtypes, output_dtypes.31 input_dtypes, output_dtypes.
31- '''32+ """
32 import torch33 import torch
33- 34+ 
34 ori_dtype = kwargs.get("input_dtypes", ["float32"])[0]35 ori_dtype = kwargs.get("input_dtypes", ["float32"])[0]
35 x_dtype = input_x.dtype36 x_dtype = input_x.dtype
36- 37+ 
37 if ori_dtype and "bfloat16" in str(ori_dtype).lower():38 if ori_dtype and "bfloat16" in str(ori_dtype).lower():
38 x_tensor = torch.from_numpy(input_x.astype(np.float32))39 x_tensor = torch.from_numpy(input_x.astype(np.float32))
39 output = torch.trunc(x_tensor)40 output = torch.trunc(x_tensor)
@@ -45,4 +46,15 @@ def trunc_golden(input_x, **kwargs):
45 else:46 else:
46 x_tensor = torch.from_numpy(input_x)47 x_tensor = torch.from_numpy(input_x)
47 output = torch.trunc(x_tensor)48 output = torch.trunc(x_tensor)
48- return output.numpy()49+ return output.numpy()
50+ 
51+ 
52+def aclnn_trunc_golden(self, out=None, **kwargs):
53+ """
54+ Aclnn golden for aclnnTrunc.
55+ Parameters follow @aclnnTruncGetWorkspaceSize without workspaceSize & executor.
56+ All the input Tensors are torch.Tensor.
57+ """
58+ import torch
59+ 
60+ return [torch.trunc(self)]
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,absolute_precision
2+Trunc_float32_ND_fuzz_1,aclnnTrunc,"('float32', 'float32')","('ND',)","((196, 2, 1, 76),(196, 2, 1, 76))","((-1000, -10),)","(-1,)","('float32',)",0.0001
3+Trunc_float16_ND_fuzz_2,aclnnTrunc,"('float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('float16',)",0.001
4+Trunc_float32_ND_fuzz_3,aclnnTrunc,"('float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)","('float32',)",0.0001
5+Trunc_float16_ND_fuzz_4,aclnnTrunc,"('float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('float16',)",0.001
6+Trunc_float16_ND_fuzz_5,aclnnTrunc,"('float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)","('float16',)",0.001
@@ -0,0 +1,239 @@
1+#!/usr/bin/env python3
2+# -*- coding: UTF-8 -*-
3+# ----------------------------------------------------------------------------
4+# Copyright (c) 2026 Huawei Technologies Co., Ltd.
5+# This program is free software, you can redistribute it and/or modify it under the terms and conditions of
6+# CANN Open Software License Agreement Version 2.0 (the "License").
7+# Please refer to the License for details. You may not use this file except in compliance with the License.
8+# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
9+# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
10+# See LICENSE in the root of the software repository for the full text of the License.
11+# ----------------------------------------------------------------------------
12+import numpy as np
13+import torch
14+from functools import reduce
15+ 
16+PHILOX_W32_0 = 0x9E3779B9
17+PHILOX_W32_1 = 0xBB67AE85
18+PHILOX_M4x32_0 = 0xD2511F53
19+PHILOX_M4x32_1 = 0xCD9E8D57
20+CURAND_2POW32_INV = 2.3283064e-10
21+CURAND_2POW32_INV_2PI = 2.3283064e-10 * 6.2831855
22+MAX_THREADS_PER_PROCESSOR = 2048
23+MAX_BLOCK_NUMS = 2147483647
24+PHILOX_BLOCK_THREAD = 512
25+STEP = 4
26+ 
27+ 
28+def mulhilo32(a, b):
29+ product = np.uint64(a) * b.astype(np.uint64)
30+ hi = (product >> 32) & 0xFFFFFFFF
31+ lo = product & 0xFFFFFFFF
32+ return hi.astype(np.uint32), lo.astype(np.uint32)
33+ 
34+ 
35+def _philox4x32round(ctr, key):
36+ hi0, lo0 = mulhilo32(PHILOX_M4x32_0, ctr[0])
37+ hi1, lo1 = mulhilo32(PHILOX_M4x32_1, ctr[2])
38+ return np.array(
39+ [hi1 ^ ctr[1] ^ key[0], lo1, hi0 ^ ctr[3] ^ key[1], lo0], dtype=np.uint32
40+ )
41+ 
42+ 
43+def rand_Philox4x32_10(c, k):
44+ k = k.copy()
45+ for i in range(9):
46+ c = _philox4x32round(c, k)
47+ k[0] += PHILOX_W32_0
48+ k[1] += PHILOX_W32_1
49+ k[0] &= 0xFFFFFFFF
50+ k[1] &= 0xFFFFFFFF
51+ return _philox4x32round(c, k)
52+ 
53+ 
54+class RandStatePhilox4_32_10:
55+ def __init__(self):
56+ self.ctr = np.array([0, 0, 0, 0], dtype=np.uint32)
57+ self.key = np.array([0, 0], dtype=np.uint32)
58+ self.STATE = 0
59+ self.output = np.array([0, 0, 0, 0], dtype=np.uint32)
60+ 
61+ 
62+def Philox_State_Incr_hi(s, n):
63+ nlo = n & 0xFFFFFFFF
64+ nhi = (n >> 32) & 0xFFFFFFFF
65+ s.ctr[2] += nlo
66+ if s.ctr[2] < nlo:
67+ nhi += 1
68+ s.ctr[3] += nhi
69+ s.ctr[3] &= 0xFFFFFFFF
70+ 
71+ 
72+def Philox_State_Incr(s, n=None):
73+ if n is None:
74+ for i in range(STEP):
75+ s.ctr[i] += 1
76+ if s.ctr[i] != 0:
77+ break
78+ else:
79+ nlo = n & 0xFFFFFFFF
80+ nhi = (n >> 32) & 0xFFFFFFFF
81+ s.ctr[0] += nlo
82+ if s.ctr[0] < nlo:
83+ nhi += 1
84+ s.ctr[1] += nhi
85+ if s.ctr[1] < nhi:
86+ nhi += 1
87+ s.ctr[2] += nhi
88+ if s.ctr[2] < nhi:
89+ nhi += 1
90+ s.ctr[3] += nhi
91+ s.ctr &= 0xFFFFFFFF
92+ 
93+ 
94+def skipahead_sequence(n, state):
95+ Philox_State_Incr_hi(state, n)
96+ state.output = rand_Philox4x32_10(state.ctr, state.key)
97+ 
98+ 
99+def skipahead(n, state):
100+ state.STATE += n % 4
101+ n //= 4
102+ if state.STATE > 3:
103+ n += 1
104+ state.STATE -= 4
105+ Philox_State_Incr(state, n)
106+ state.output = rand_Philox4x32_10(state.ctr, state.key)
107+ 
108+ 
109+def rand_init(seed, subsequence, offset, state):
110+ state.ctr = np.array([0, 0, 0, 0], dtype=np.uint32)
111+ state.key[0] = seed & 0xFFFFFFFF
112+ state.key[1] = (seed >> 32) & 0xFFFFFFFF
113+ state.STATE = 0
114+ skipahead_sequence(subsequence, state)
115+ skipahead(offset, state)
116+ 
117+ 
118+def curand(state):
119+ r = state.output[state.STATE]
120+ state.STATE += 1
121+ if state.STATE == 4:
122+ Philox_State_Incr(state)
123+ state.output = rand_Philox4x32_10(state.ctr, state.key)
124+ state.STATE = 0
125+ return r
126+ 
127+ 
128+def rand4(state):
129+ tmp = state.output.copy()
130+ Philox_State_Incr(state)
131+ state.output = rand_Philox4x32_10(state.ctr, state.key)
132+ if state.STATE == 0:
133+ return tmp
134+ r = np.zeros(STEP, dtype=np.uint32)
135+ if state.STATE == 1:
136+ r[0], r[1], r[2], r[3] = tmp[1], tmp[2], tmp[3], state.output[0]
137+ elif state.STATE == 2:
138+ r[0], r[1], r[2], r[3] = tmp[2], tmp[3], state.output[0], state.output[1]
139+ elif state.STATE == 3:
140+ r[0], r[1], r[2], r[3] = (
141+ tmp[3],
142+ state.output[0],
143+ state.output[1],
144+ state.output[2],
145+ )
146+ return r
147+ 
148+ 
149+def rand_uniform4(state):
150+ x = rand4(state)
151+ y = np.array(
152+ [
153+ x[0] * CURAND_2POW32_INV + (CURAND_2POW32_INV / 2),
154+ x[1] * CURAND_2POW32_INV + (CURAND_2POW32_INV / 2),
155+ x[2] * CURAND_2POW32_INV + (CURAND_2POW32_INV / 2),
156+ x[3] * CURAND_2POW32_INV + (CURAND_2POW32_INV / 2),
157+ ],
158+ dtype=np.float32,
159+ )
160+ return y
161+ 
162+ 
163+def bernhoulli_h20(seed, offset, prob, total_ele):
164+ result = np.zeros(total_ele, dtype=np.uint32)
165+ minBlockNums = (
166+ total_ele + MAX_THREADS_PER_PROCESSOR - 1
167+ ) // MAX_THREADS_PER_PROCESSOR
168+ minBlockNums = min(minBlockNums, MAX_BLOCK_NUMS)
169+ total_thread = minBlockNums * PHILOX_BLOCK_THREAD
170+ repeat_time = ((total_ele + STEP - 1) // STEP + total_thread - 1) // total_thread
171+ for i in range(total_thread):
172+ s = RandStatePhilox4_32_10()
173+ rand_init(seed, i, offset, s)
174+ for j in range(repeat_time):
175+ x = rand_uniform4(s)
176+ realIndex = i * STEP + total_thread * STEP * j
177+ if realIndex >= total_ele:
178+ break
179+ result[realIndex] = 1 if x[0] <= prob[realIndex] else 0
180+ if realIndex + 1 >= total_ele:
181+ break
182+ result[realIndex + 1] = 1 if x[1] <= prob[realIndex + 1] else 0
183+ if realIndex + 2 >= total_ele:
184+ break
185+ result[realIndex + 2] = 1 if x[2] <= prob[realIndex + 2] else 0
186+ if realIndex + 3 >= total_ele:
187+ break
188+ result[realIndex + 3] = 1 if x[3] <= prob[realIndex + 3] else 0
189+ return result
190+ 
191+ 
192+class _NumpyBFloat16:
193+ def __new__(cls):
194+ return np.dtype("float32")
195+ 
196+ 
197+def numpy_bfloat16():
198+ return np.dtype("float32")
199+ 
200+ 
201+__golden__ = {
202+ "aclnn": {
203+ "aclnnBernoulli": "aclnn_bernoulli_golden",
204+ }
205+}
206+ 
207+ 
208+def aclnn_bernoulli_golden(self, prob, seed=0, offset=0, out=None, **kwargs):
209+ """
210+ Aclnn golden for aclnnBernoulli.
211+ Parameters follow @aclnnBernoulliGetWorkspaceSize without workspaceSize & executor.
212+ All the input Tensors are torch.Tensor.
213+ """
214+ if hasattr(prob, "item"):
215+ prob_val = prob.item()
216+ else:
217+ prob_val = prob
218+ if hasattr(seed, "item"):
219+ seed = seed.item()
220+ if hasattr(offset, "item"):
221+ offset = offset.item()
222+ 
223+ seed = np.int64(seed).astype(np.uint64)
224+ offset = np.int64(offset).astype(np.uint64)
225+ 
226+ x_shape = list(self.shape)
227+ tol = reduce(lambda x, y: x * y, x_shape) if x_shape else 1
228+ if tol == 0:
229+ return [torch.empty(x_shape), torch.empty(x_shape)]
230+ 
231+ prob_arr = np.array([prob_val])
232+ out_total_ele = int(reduce(lambda x, y: x * y, x_shape)) if x_shape else 1
233+ if prob_arr.size == 1:
234+ prob_arr = np.broadcast_to(prob_arr, out_total_ele)
235+ 
236+ result = bernhoulli_h20(
237+ seed=seed, offset=offset, prob=prob_arr, total_ele=out_total_ele
238+ )
239+ return [torch.from_numpy(result).to(out.dtype)]
@@ -0,0 +1,6 @@
1+testcase_name,api_name,tensor_view_shapes,tensor_formats,tensor_dtypes,scalar_dtypes,attributes,output_tensor_indexes,precision_tolerances,absolute_precision,is_enabled
2+aclnn_bernoulli_test_0000,aclnnBernoulli,"((6,),(6,))","(('ND', 'ND'))","('uint8', 'uint8')","('float32',)","{'prob': 0.5, 'seed': 67, 'offset': 256}","(-1,)","((0.0001, 0.0001),)",0,TRUE
3+aclnn_bernoulli_test_0001,aclnnBernoulli,"((1, 1, 1, 1, 1, 1),(1, 1, 1, 1, 1, 1))","(('ND', 'ND'))","('float16', 'float16')","('float32',)","{'prob': 0.12345, 'seed': 1, 'offset': 400}","(-1,)","((0.001, 0.001),)",0,TRUE
4+aclnn_bernoulli_test_0002,aclnnBernoulli,"((100, 101, 102),(100, 101, 102))","(('ND', 'ND'))","('float32', 'float32')","('float16',)","{'prob': 0.12345, 'seed': 1, 'offset': 128}","(-1,)","((0.0001, 0.0001),)",0,TRUE
5+aclnn_bernoulli_test_0003,aclnnBernoulli,"((1024, 16, 8),(1024, 16, 8))","(('ND', 'ND'))","('int8', 'int8')","('float16',)","{'prob': 0.1111, 'seed': 100, 'offset': 64}","(-1,)","((0.001, 0.001),)",0,TRUE
6+aclnn_bernoulli_test_0004,aclnnBernoulli,"((0, 16, 8, 1024),(0, 16, 8, 1024))","(('ND', 'ND'))","('int8', 'int8')","('float32',)","{'prob': 0.1, 'seed': 100, 'offset': 512}","(-1,)","((0.0001, 0.0001),)",0,TRUE