已合并
feat: Add aclnn st cases and golden functions #5200
yanzhi2024创建于 8月29日
feat: Add aclnn st cases and golden functions #5200
已合并
共 66 个文件变更+1196-291
| @@ -12,25 +12,32 @@ | |||
| 12 | 12 | ||
| 13 | import numpy | 13 | import numpy |
| 14 | 14 | ||
| 15 | +import torch | ||
| 16 | + | ||
| 15 | __golden__ = { | 17 | __golden__ = { |
| 16 | - "kernel": { | 18 | + "aclnn": { |
| 17 | - "concat_d": "concat_d_golden" | 19 | + "aclnnCat": "aclnn_cat_golden", |
| 18 | - } | 20 | + }, |
| 21 | + "kernel": {"concat_d": "concat_d_golden"}, | ||
| 19 | } | 22 | } |
| 20 | 23 | ||
| 21 | 24 | ||
| 22 | -def update_axis_for_hw_inner_format(ori_shape, axis, input_format, ori_format, reduce_mode=False): | 25 | +def update_axis_for_hw_inner_format( |
| 26 | + ori_shape, axis, input_format, ori_format, reduce_mode=False | ||
| 27 | +): | ||
| 23 | if input_format in ("NDC1HWC0", "NC1HWC0"): | 28 | if input_format in ("NDC1HWC0", "NC1HWC0"): |
| 24 | ori_shape_len = len(ori_shape) if -2 not in ori_shape else len(ori_format) | 29 | ori_shape_len = len(ori_shape) if -2 not in ori_shape else len(ori_format) |
| 25 | axis = axis % ori_shape_len | 30 | axis = axis % ori_shape_len |
| 26 | offset_6hd = 1 if input_format == "NDC1HWC0" else 0 | 31 | offset_6hd = 1 if input_format == "NDC1HWC0" else 0 |
| 27 | - format_c_axis = 1 + offset_6hd if not reduce_mode else [1 + offset_6hd, 4 + offset_6hd] | 32 | + format_c_axis = ( |
| 33 | + 1 + offset_6hd if not reduce_mode else [1 + offset_6hd, 4 + offset_6hd] | ||
| 34 | + ) | ||
| 28 | format_axis_map = { | 35 | format_axis_map = { |
| 29 | "N": 0, | 36 | "N": 0, |
| 30 | "C": format_c_axis, | 37 | "C": format_c_axis, |
| 31 | "H": 2 + offset_6hd, | 38 | "H": 2 + offset_6hd, |
| 32 | "W": 3 + offset_6hd, | 39 | "W": 3 + offset_6hd, |
| 33 | - "D": 1 | 40 | + "D": 1, |
| 34 | } | 41 | } |
| 35 | concat_dim_name = ori_format[axis] | 42 | concat_dim_name = ori_format[axis] |
| 36 | axis = format_axis_map[concat_dim_name] | 43 | axis = format_axis_map[concat_dim_name] |
| @@ -38,21 +45,33 @@ def update_axis_for_hw_inner_format(ori_shape, axis, input_format, ori_format, r | |||
| 38 | if input_format in ("FRACTAL_NZ",): | 45 | if input_format in ("FRACTAL_NZ",): |
| 39 | axis = axis % len(ori_shape) | 46 | axis = axis % len(ori_shape) |
| 40 | if axis == len(ori_shape) - 1: | 47 | if axis == len(ori_shape) - 1: |
| 41 | - axis = len(ori_shape) - 2 if not reduce_mode else [len(ori_shape) - 2, len(ori_shape) + 1] | 48 | + axis = ( |
| 49 | + len(ori_shape) - 2 | ||
| 50 | + if not reduce_mode | ||
| 51 | + else [len(ori_shape) - 2, len(ori_shape) + 1] | ||
| 52 | + ) | ||
| 42 | elif axis == len(ori_shape) - 2: | 53 | elif axis == len(ori_shape) - 2: |
| 43 | - axis = len(ori_shape) - 1 if not reduce_mode else [len(ori_shape) - 1, len(ori_shape) + 0] | 54 | + axis = ( |
| 55 | + len(ori_shape) - 1 | ||
| 56 | + if not reduce_mode | ||
| 57 | + else [len(ori_shape) - 1, len(ori_shape) + 0] | ||
| 58 | + ) | ||
| 44 | 59 | ||
| 45 | if input_format in ("FRACTAL_Z", "FRACTAL_Z_3D"): | 60 | if input_format in ("FRACTAL_Z", "FRACTAL_Z_3D"): |
| 46 | axis = axis % len(ori_shape) | 61 | axis = axis % len(ori_shape) |
| 47 | offset_3d = 1 if input_format == "FRACTAL_Z_3D" else 0 | 62 | offset_3d = 1 if input_format == "FRACTAL_Z_3D" else 0 |
| 48 | - format_c_axis = 0 + offset_3d if not reduce_mode else [0 + offset_3d, 5 + offset_3d] | 63 | + format_c_axis = ( |
| 49 | - format_n_axis = 3 + offset_3d if not reduce_mode else [3 + offset_3d, 4 + offset_3d] | 64 | + 0 + offset_3d if not reduce_mode else [0 + offset_3d, 5 + offset_3d] |
| 65 | + ) | ||
| 66 | + format_n_axis = ( | ||
| 67 | + 3 + offset_3d if not reduce_mode else [3 + offset_3d, 4 + offset_3d] | ||
| 68 | + ) | ||
| 50 | format_axis_map = { | 69 | format_axis_map = { |
| 51 | "N": format_n_axis, | 70 | "N": format_n_axis, |
| 52 | "C": format_c_axis, | 71 | "C": format_c_axis, |
| 53 | "H": 1 + offset_3d, | 72 | "H": 1 + offset_3d, |
| 54 | "W": 2 + offset_3d, | 73 | "W": 2 + offset_3d, |
| 55 | - "D": 0 | 74 | + "D": 0, |
| 56 | } | 75 | } |
| 57 | concat_dim_name = ori_format[axis] | 76 | concat_dim_name = ori_format[axis] |
| 58 | axis = format_axis_map[concat_dim_name] | 77 | axis = format_axis_map[concat_dim_name] |
| @@ -61,7 +80,7 @@ def update_axis_for_hw_inner_format(ori_shape, axis, input_format, ori_format, r | |||
| 61 | 80 | ||
| 62 | 81 | ||
| 63 | def concat_d_golden(x, *, concat_dim, N=1, **kwargs): | 82 | def concat_d_golden(x, *, concat_dim, N=1, **kwargs): |
| 64 | - ''' | 83 | + """ |
| 65 | Golden function for concat. | 84 | Golden function for concat. |
| 66 | All the parameters (names and order) follow @concat_d_def.cpp without outputs. | 85 | All the parameters (names and order) follow @concat_d_def.cpp without outputs. |
| 67 | All the input Tensors are numpy.ndarray. | 86 | All the input Tensors are numpy.ndarray. |
| @@ -72,12 +91,20 @@ def concat_d_golden(x, *, concat_dim, N=1, **kwargs): | |||
| 72 | 91 | ||
| 73 | Returns: | 92 | Returns: |
| 74 | Output tensor | 93 | Output tensor |
| 75 | - ''' | 94 | + """ |
| 76 | x_arrays = list(x) | 95 | x_arrays = list(x) |
| 77 | 96 | ||
| 78 | - ori_shape = kwargs.get('input_ori_shapes', [x[0].shape])[0] | 97 | + ori_shape = kwargs.get("input_ori_shapes", [x[0].shape])[0] |
| 79 | - input_formats = kwargs.get('input_formats', ['ND']) | 98 | + input_formats = kwargs.get("input_formats", ["ND"]) |
| 80 | - input_ori_formats = kwargs.get('input_ori_formats', ['ND']) | 99 | + input_ori_formats = kwargs.get("input_ori_formats", ["ND"]) |
| 81 | - | 100 | + |
| 82 | - concat_dim = update_axis_for_hw_inner_format(ori_shape, concat_dim, input_formats[0], input_ori_formats[0]) | 101 | + concat_dim = update_axis_for_hw_inner_format( |
| 102 | + ori_shape, concat_dim, input_formats[0], input_ori_formats[0] | ||
| 103 | + ) | ||
| 83 | return numpy.concatenate(x_arrays, axis=concat_dim) | 104 | return numpy.concatenate(x_arrays, axis=concat_dim) |
| 105 | + | ||
| 106 | + | ||
| 107 | +def aclnn_cat_golden(tensors, dim, out, **kwargs): | ||
| 108 | + if hasattr(dim, "item"): | ||
| 109 | + dim = dim.item() | ||
| 110 | + return [torch.cat(tensors, dim=dim)] | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_view_shapes,tensor_dtypes,attributes,output_tensor_indexes,input_data_ranges,precision_tolerances,absolute_precision | ||
| 2 | +aclnnCat_float,aclnnCat,"(((3, 3), (3, 2)), (3, 5))","(('float32', 'float32'), 'float32')",{'dim': -1},"(1,)",,, | ||
| 3 | +aclnnCat_bf16,aclnnCat,"(((4, 5), (4, 5)), (4, 10))","(('bfloat16', 'bfloat16'), 'bfloat16')",{'dim': -1},"(1,)",,, | ||
| 4 | +aclnnCat_int8,aclnnCat,"(((13, 20), (13, 20)), (26, 20))","(('int8', 'int8'), 'int8')",{'dim': 0},"(1,)",,, | ||
| 5 | +aclnnCat_int64,aclnnCat,"(((30, 30), (30, 30)), (30, 60))","(('int64', 'int64'), 'int64')",{'dim': -1},"(1,)",,, | ||
| 6 | +aclnnCat_float16,aclnnCat,"(((3, 4), (3, 4)), (3, 8))","(('float16', 'float16'), 'float16')",{'dim': -1},"(1,)",,, | ||
| @@ -10,23 +10,30 @@ | |||
| 10 | # See LICENSE in the root of the software repository for the full text of the License. | 10 | # See LICENSE in the root of the software repository for the full text of the License. |
| 11 | # ---------------------------------------------------------------------------- | 11 | # ---------------------------------------------------------------------------- |
| 12 | import numpy as np | 12 | import numpy as np |
| 13 | +import torch | ||
| 13 | 14 | ||
| 14 | __golden__ = { | 15 | __golden__ = { |
| 15 | - "kernel": { | 16 | + "aclnn": { |
| 16 | - "pack": "pack_golden" | 17 | + "aclnnStack": "aclnn_stack_golden", |
| 17 | - } | 18 | + }, |
| 19 | + "kernel": {"pack": "pack_golden"}, | ||
| 18 | } | 20 | } |
| 19 | - | 21 | + |
| 20 | -def pack_golden(x, | 22 | + |
| 21 | - axis: int=0, N: int=1, | 23 | +def pack_golden(x, axis: int = 0, N: int = 1, **kwargs): |
| 22 | - **kwargs): | 24 | + """ |
| 23 | - ''' | ||
| 24 | Kernel golden for pack. | 25 | Kernel golden for pack. |
| 25 | All the parameters follow @pack_def.cpp without outputs. | 26 | All the parameters follow @pack_def.cpp without outputs. |
| 26 | All the input Tensors are numpy.ndarray. | 27 | All the input Tensors are numpy.ndarray. |
| 27 | - kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, | 28 | + kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, |
| 28 | - input_formats, output_formats, input_ori_formats, output_ori_formats, | 29 | + input_formats, output_formats, input_ori_formats, output_ori_formats, |
| 29 | - input_dtypes, output_dtypes. | 30 | + input_dtypes, output_dtypes. |
| 30 | - ''' | 31 | + """ |
| 31 | - | 32 | + |
| 32 | return np.stack(x, axis=axis) | 33 | return np.stack(x, axis=axis) |
| 34 | + | ||
| 35 | + | ||
| 36 | +def aclnn_stack_golden(tensors, dim, out, **kwargs): | ||
| 37 | + if hasattr(dim, "item"): | ||
| 38 | + dim = dim.item() | ||
| 39 | + return [torch.stack(tensors, dim=dim)] | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_view_shapes,tensor_dtypes,attributes,output_tensor_indexes,input_data_ranges,precision_tolerances,absolute_precision | ||
| 2 | +aclnnStack_float,aclnnStack,"(((3, 3), (3, 3)), (3, 3, 2))","(('float32', 'float32'), 'float32')",{'dim': -1},"(1,)",,, | ||
| 3 | +aclnnStack_bf16,aclnnStack,"(((4, 5), (4, 5)), (4, 5, 2))","(('bfloat16', 'bfloat16'), 'bfloat16')",{'dim': -1},"(1,)",,, | ||
| 4 | +aclnnStack_int8,aclnnStack,"(((13, 20), (13, 20)), (2, 13, 20))","(('int8', 'int8'), 'int8')",{'dim': 0},"(1,)",,, | ||
| 5 | +aclnnStack_int64,aclnnStack,"(((30, 30), (30, 30)), (30, 30, 2))","(('int64', 'int64'), 'int64')",{'dim': -1},"(1,)",,, | ||
| 6 | +aclnnStack_float16,aclnnStack,"(((3, 4), (3, 4)), (3, 4, 2))","(('float16', 'float16'), 'float16')",{'dim': -1},"(1,)",,, | ||
| @@ -11,23 +11,33 @@ | |||
| 11 | # ---------------------------------------------------------------------------- | 11 | # ---------------------------------------------------------------------------- |
| 12 | 12 | ||
| 13 | import numpy as np | 13 | import numpy as np |
| 14 | +import torch | ||
| 14 | 15 | ||
| 15 | 16 | ||
| 16 | __golden__ = { | 17 | __golden__ = { |
| 17 | - "kernel": { | 18 | + "aclnn": { |
| 18 | - "transpose": "transpose_golden" | 19 | + "aclnnPermute": "aclnn_permute_golden", |
| 19 | - } | 20 | + }, |
| 21 | + "kernel": {"transpose": "transpose_golden"}, | ||
| 20 | } | 22 | } |
| 21 | 23 | ||
| 22 | 24 | ||
| 23 | def transpose_golden(x, perm, **kwargs): | 25 | def transpose_golden(x, perm, **kwargs): |
| 24 | - ''' | 26 | + """ |
| 25 | Kernel golden for transpose / transpose_d. | 27 | Kernel golden for transpose / transpose_d. |
| 26 | All the parameters follow @transpose_def.cpp without outputs. | 28 | All the parameters follow @transpose_def.cpp without outputs. |
| 27 | All the input Tensors are numpy.ndarray. | 29 | All the input Tensors are numpy.ndarray. |
| 28 | kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, | 30 | kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, |
| 29 | input_formats, output_formats, input_ori_formats, output_ori_formats, | 31 | input_formats, output_formats, input_ori_formats, output_ori_formats, |
| 30 | input_dtypes, output_dtypes. | 32 | input_dtypes, output_dtypes. |
| 31 | - ''' | 33 | + """ |
| 32 | perm_val = perm.tolist() if isinstance(perm, np.ndarray) else perm | 34 | perm_val = perm.tolist() if isinstance(perm, np.ndarray) else perm |
| 33 | return np.transpose(x, perm_val) | 35 | return np.transpose(x, perm_val) |
| 36 | + | ||
| 37 | + | ||
| 38 | +def aclnn_permute_golden(self, dims=0, out=None, **kwargs): | ||
| 39 | + if hasattr(dims, "tolist"): | ||
| 40 | + dims = dims.tolist() | ||
| 41 | + elif isinstance(dims, int): | ||
| 42 | + dims = [dims] | ||
| 43 | + return [torch.permute(self, dims)] | ||
| @@ -0,0 +1,2 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,attributes,tensor_view_shapes,output_tensor_indexes,input_data_ranges,precision_tolerances,absolute_precision | ||
| 2 | +aclnnPermute_00,aclnnPermute,"('float16', 'float16')","{'dims': [2,1,0]}","((256,64,128),(128,64,256))","(1,)","[[0, 0.001]]","((0.001, 0.001),)",0.000001 | ||
| @@ -0,0 +1,22 @@ | |||
| 1 | +#!/usr/bin/env python3 | ||
| 2 | +# -*- coding: UTF-8 -*- | ||
| 3 | +# ---------------------------------------------------------------------------- | ||
| 4 | +# Copyright (c) 2026 Huawei Technologies Co., Ltd. | ||
| 5 | +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of | ||
| 6 | +# CANN Open Software License Agreement Version 2.0 (the "License"). | ||
| 7 | +# Please refer to the License for details. You may not use this file except in compliance with the License. | ||
| 8 | +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, | ||
| 9 | +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. | ||
| 10 | +# See LICENSE in the root of the software repository for the full text of the License. | ||
| 11 | +# ---------------------------------------------------------------------------- | ||
| 12 | + | ||
| 13 | +__golden__ = { | ||
| 14 | + "aclnn": { | ||
| 15 | + "aclnnInplaceCopy": "aclnn_inplace_copy_golden", | ||
| 16 | + } | ||
| 17 | +} | ||
| 18 | + | ||
| 19 | + | ||
| 20 | +def aclnn_inplace_copy_golden(selfRef, src, **kwargs): | ||
| 21 | + selfRef.copy_(src) | ||
| 22 | + return [selfRef] | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,attributes,tensor_view_shapes,output_tensor_indexes,tensor_storage_shapes,tensor_view_strides,tensor_view_offsets,input_data_ranges | ||
| 2 | +aclnnInplaceCopy_random_000001,aclnnInplaceCopy,"('float32', 'float32')","('ND', 'ND')","{'kernelShape': [2,2], 'strides': [2,2], 'autoPad' :0, 'pads' : [0], 'dilations': [1], 'ceilMode':False}","((4,2), (4,2))","(0,)","((4,8), (4,2))","((8,1), (2,1))","(0, 0,)","((0.001, 0.01),)" | ||
| 3 | +aclnnInplaceCopy_random_000002,aclnnInplaceCopy,"('float16', 'float16')","('ND', 'ND')","{'kernelShape': [2,2], 'strides': [2,2], 'autoPad' :0, 'pads' : [0], 'dilations': [1], 'ceilMode':False}","((2,10), (2,10))","(0,)","((2,64), (2,10))","((64,1), (10,1))","(0, 0,)","((-1000, -10),)" | ||
| 4 | +aclnnInplaceCopy_random_000003,aclnnInplaceCopy,"('bfloat16', 'bfloat16')","('ND', 'ND')","{'kernelShape': [2,2], 'strides': [2,2], 'autoPad' :0, 'pads' : [0], 'dilations': [1], 'ceilMode':False}","((6,40), (6,40))","(0,)","((6,64), (6,40))","((64,1), (40,1))","(0, 0,)","((1, 2),)" | ||
| 5 | +aclnnInplaceCopy_random_000004,aclnnInplaceCopy,"('float32', 'float32')","('ND', 'ND')","{'kernelShape': [2,2], 'strides': [2,2], 'autoPad' :0, 'pads' : [0], 'dilations': [1], 'ceilMode':False}","((2, 3, 5), (2, 3, 5))","(0,)","((2, 5, 5), (2, 3, 5))","((25, 5, 1), (15, 5, 1))","(0, 0,)","((-1, -0.01),)" | ||
| 6 | +aclnnInplaceCopy_random_000005,aclnnInplaceCopy,"('float32', 'float32')","('ND', 'ND')","{'kernelShape': [3], 'strides': [3], 'autoPad' :0, 'pads' : [1], 'dilations': [1], 'ceilMode':True}","((4, 5, 6), (4, 5, 6))","(0,)","((4, 10, 6), (4, 5, 6))","((60, 6, 1), (30, 6, 1))","(0, 0,)","((-0.01, 0.01),)" | ||
| @@ -41,7 +41,7 @@ def abs_golden(x, **kwargs): | |||
| 41 | return np.abs(x) | 41 | return np.abs(x) |
| 42 | 42 | ||
| 43 | 43 | ||
| 44 | -def aclnn_abs_golden(selfT, out=None, **kwargs): | 44 | +def aclnn_abs_golden(self, out=None, **kwargs): |
| 45 | """ | 45 | """ |
| 46 | Aclnn golden for aclnnAbs. | 46 | Aclnn golden for aclnnAbs. |
| 47 | Parameters follow @aclnnAbsGetWorkspaceSize without workspaceSize & executor. | 47 | Parameters follow @aclnnAbsGetWorkspaceSize without workspaceSize & executor. |
| @@ -50,4 +50,4 @@ def aclnn_abs_golden(selfT, out=None, **kwargs): | |||
| 50 | kwargs may contain: tensor_dtypes, tensor_formats, scalar_dtypes, | 50 | kwargs may contain: tensor_dtypes, tensor_formats, scalar_dtypes, |
| 51 | use_torch, short_soc_version, testcase_name. | 51 | use_torch, short_soc_version, testcase_name. |
| 52 | """ | 52 | """ |
| 53 | - return torch.abs(selfT) | 53 | + return torch.abs(self) |
| @@ -1,14 +1,6 @@ | |||
| 1 | -testcase_name,api_name,tensor_view_shapes,tensor_dtypes,tensor_formats,attributes,output_tensor_indexes,input_data_ranges,absolute_precision | 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,absolute_precision |
| 2 | -Abs_float32_ND_fuzz_1,aclnnAbs,"((196, 2, 1, 76, 1, 1), (196, 2, 1, 76, 1, 1))","('float32', 'float32')","('ND',)",{},"(-1,)","((-1000, -10),)",0.0001 | 2 | +Abs_float32_ND_fuzz_1,aclnnAbs,"('float32', 'float32')","('ND',)","((196, 2, 1, 76, 1, 1), (196, 2, 1, 76, 1, 1))","((-1000, -10),)","(-1,)",0.0001 |
| 3 | -Abs_float16_ND_fuzz_2,aclnnAbs,"((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","('float16', 'float16')","('ND',)",{},"(-1,)","((0, 0),)",0.001 | 3 | +Abs_float16_ND_fuzz_2,aclnnAbs,"('float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)",0.001 |
| 4 | -Abs_float32_ND_fuzz_3,aclnnAbs,"((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","('float32', 'float32')","('ND',)",{},"(-1,)","((-3.4e+38, 3.4e+38),)",0.0001 | 4 | +Abs_float32_ND_fuzz_3,aclnnAbs,"('float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)",0.0001 |
| 5 | -Abs_float16_ND_fuzz_4,aclnnAbs,"((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","('float16', 'float16')","('ND',)",{},"(-1,)","((-10, -2),)",0.001 | 5 | +Abs_float16_ND_fuzz_4,aclnnAbs,"('float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)",0.001 |
| 6 | -Abs_float16_ND_fuzz_5,aclnnAbs,"((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","('float16', 'float16')","('ND',)",{},"(-1,)","((-1, 1),)",0.001 | 6 | +Abs_float16_ND_fuzz_5,aclnnAbs,"('float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)",0.001 |
| 7 | -Abs_float32_ND_fuzz_6,aclnnAbs,"((1, 64, 10, 1, 192, 1), (1, 64, 10, 1, 192, 1))","('float32', 'float32')","('ND',)",{},"(-1,)","((-1, 1),)",0.0001 | ||
| 8 | -Abs_bfloat16_ND_fuzz_7,aclnnAbs,"((1, 64, 10, 1, 192, 1), (1, 64, 10, 1, 192, 1))","('bfloat16', 'bfloat16')","('ND',)",{},"(-1,)","((-1, 1),)",0.0001 | ||
| 9 | -Abs_int8_ND_fuzz_8,aclnnAbs,"((1, 64, 10, 1, 192, 1), (1, 64, 10, 1, 192, 1))","('int8', 'int8')","('ND',)",{},"(-1,)","((-1, 1),)",0.0001 | ||
| 10 | -Abs_int32_ND_fuzz_9,aclnnAbs,"((1, 64, 10, 1, 192, 1), (1, 64, 10, 1, 192, 1))","('int32', 'int32')","('ND',)",{},"(-1,)","((-1, 1),)",0.0001 | ||
| 11 | -Abs_int64_ND_fuzz_10,aclnnAbs,"((1, 64, 10, 1, 192, 1), (1, 64, 10, 1, 192, 1))","('int64', 'int64')","('ND',)",{},"(-1,)","((-1, 1),)",0.0001 | ||
| 12 | -Abs_int8_ND_fuzz_11,aclnnAbs,"((1, 64, 10, 1, 192, 1), (1, 64, 10, 1, 192, 1))","('int8', 'int8')","('ND',)",{},"(-1,)","((-1, 1),)",0.0001 | ||
| 13 | -Abs_int16_ND_fuzz_12,aclnnAbs,"((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","('int16', 'int16')","('ND',)",{},"(-1,)","((-1, 1),)",0.001 | ||
| 14 | - | ||
| @@ -9,84 +9,98 @@ | |||
| 9 | # INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. | 9 | # INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. |
| 10 | # See LICENSE in the root of the software repository for the full text of the License. | 10 | # See LICENSE in the root of the software repository for the full text of the License. |
| 11 | # ---------------------------------------------------------------------------- | 11 | # ---------------------------------------------------------------------------- |
| 12 | - | 12 | + |
| 13 | import numpy as np | 13 | import numpy as np |
| 14 | - | 14 | + |
| 15 | __golden__ = { | 15 | __golden__ = { |
| 16 | - "kernel": { | 16 | + "aclnn": { |
| 17 | - "cast": "cast_golden" | 17 | + "aclnnCast": "aclnn_cast_golden", |
| 18 | - } | 18 | + }, |
| 19 | + "kernel": {"cast": "cast_golden"}, | ||
| 19 | } | 20 | } |
| 20 | - | 21 | + |
| 21 | _DATA_TYPE_INT_TO_STR = { | 22 | _DATA_TYPE_INT_TO_STR = { |
| 22 | - 0: 'float32', | 23 | + 0: "float32", |
| 23 | - 1: 'float16', | 24 | + 1: "float16", |
| 24 | - 2: 'int8', | 25 | + 2: "int8", |
| 25 | - 3: 'int32', | 26 | + 3: "int32", |
| 26 | - 4: 'uint8', | 27 | + 4: "uint8", |
| 27 | - 6: 'int16', | 28 | + 6: "int16", |
| 28 | - 7: 'uint16', | 29 | + 7: "uint16", |
| 29 | - 8: 'uint32', | 30 | + 8: "uint32", |
| 30 | - 9: 'int64', | 31 | + 9: "int64", |
| 31 | - 10: 'uint64', | 32 | + 10: "uint64", |
| 32 | - 11: 'double', | 33 | + 11: "double", |
| 33 | - 12: 'bool', | 34 | + 12: "bool", |
| 34 | - 16: 'complex64', | 35 | + 16: "complex64", |
| 35 | - 17: 'complex128', | 36 | + 17: "complex128", |
| 36 | - 27: 'bfloat16', | 37 | + 27: "bfloat16", |
| 37 | - 29: 'int4', | 38 | + 29: "int4", |
| 38 | - 30: 'uint1', | 39 | + 30: "uint1", |
| 39 | - 33: 'complex32', | 40 | + 33: "complex32", |
| 40 | - 34: 'hifloat8', | 41 | + 34: "hifloat8", |
| 41 | - 35: 'float8_e5m2', | 42 | + 35: "float8_e5m2", |
| 42 | - 36: 'float8_e4m3fn', | 43 | + 36: "float8_e4m3fn", |
| 43 | - 40: 'float4_e2m1', | 44 | + 40: "float4_e2m1", |
| 44 | - 41: 'float4_e1m2', | 45 | + 41: "float4_e1m2", |
| 45 | } | 46 | } |
| 46 | - | 47 | + |
| 47 | -_SPECIAL_DTYPES = ("bfloat16", "int4", | 48 | +_SPECIAL_DTYPES = ( |
| 48 | - "float8_e5m2", "float8_e4m3fn", | 49 | + "bfloat16", |
| 49 | - "float4_e2m1", "float4_e1m2", | 50 | + "int4", |
| 50 | - "hifloat8") | 51 | + "float8_e5m2", |
| 51 | - | 52 | + "float8_e4m3fn", |
| 53 | + "float4_e2m1", | ||
| 54 | + "float4_e1m2", | ||
| 55 | + "hifloat8", | ||
| 56 | +) | ||
| 57 | + | ||
| 58 | + | ||
| 52 | def _resolve_custom_numpy_dtype(dtype_str): | 59 | def _resolve_custom_numpy_dtype(dtype_str): |
| 53 | if dtype_str == "bfloat16": | 60 | if dtype_str == "bfloat16": |
| 54 | from ml_dtypes import bfloat16 | 61 | from ml_dtypes import bfloat16 |
| 62 | + | ||
| 55 | return bfloat16 | 63 | return bfloat16 |
| 56 | elif dtype_str == "int4": | 64 | elif dtype_str == "int4": |
| 57 | from ml_dtypes import int4 | 65 | from ml_dtypes import int4 |
| 66 | + | ||
| 58 | return int4 | 67 | return int4 |
| 59 | elif dtype_str == "float8_e5m2": | 68 | elif dtype_str == "float8_e5m2": |
| 60 | from ml_dtypes import float8_e5m2 | 69 | from ml_dtypes import float8_e5m2 |
| 70 | + | ||
| 61 | return float8_e5m2 | 71 | return float8_e5m2 |
| 62 | elif dtype_str == "float8_e4m3fn": | 72 | elif dtype_str == "float8_e4m3fn": |
| 63 | from ml_dtypes import float8_e4m3fn | 73 | from ml_dtypes import float8_e4m3fn |
| 74 | + | ||
| 64 | return float8_e4m3fn | 75 | return float8_e4m3fn |
| 65 | elif dtype_str == "hifloat8": | 76 | elif dtype_str == "hifloat8": |
| 66 | from en_dtypes import hifloat8 | 77 | from en_dtypes import hifloat8 |
| 78 | + | ||
| 67 | return hifloat8 | 79 | return hifloat8 |
| 68 | elif dtype_str == "float4_e2m1": | 80 | elif dtype_str == "float4_e2m1": |
| 69 | from ml_dtypes import float4_e2m1 | 81 | from ml_dtypes import float4_e2m1 |
| 82 | + | ||
| 70 | return float4_e2m1 | 83 | return float4_e2m1 |
| 71 | elif dtype_str == "float4_e1m2": | 84 | elif dtype_str == "float4_e1m2": |
| 72 | from ml_dtypes import float4_e1m2 | 85 | from ml_dtypes import float4_e1m2 |
| 86 | + | ||
| 73 | return float4_e1m2 | 87 | return float4_e1m2 |
| 74 | return None | 88 | return None |
| 75 | - | 89 | + |
| 76 | -def cast_golden(x, | 90 | + |
| 77 | - dst_type: int, | 91 | +def cast_golden(x, dst_type: int, **kwargs): |
| 78 | - **kwargs): | 92 | + """ |
| 79 | - ''' | ||
| 80 | Kernel golden for cast. | 93 | Kernel golden for cast. |
| 81 | All the parameters follow @cast_def.cpp without outputs. | 94 | All the parameters follow @cast_def.cpp without outputs. |
| 82 | All the input Tensors are numpy.ndarray. | 95 | All the input Tensors are numpy.ndarray. |
| 83 | kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, | 96 | kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, |
| 84 | - input_formats, output_formats, input_ori_formats, output_ori_formats, | 97 | + input_formats, output_formats, input_ori_formats, output_ori_formats, |
| 85 | - input_dtypes, output_dtypes. | 98 | + input_dtypes, output_dtypes. |
| 86 | - ''' | 99 | + """ |
| 87 | dst_type_str = _DATA_TYPE_INT_TO_STR.get(dst_type, str(dst_type)) | 100 | dst_type_str = _DATA_TYPE_INT_TO_STR.get(dst_type, str(dst_type)) |
| 88 | - if (x.dtype.name == "bfloat16" and dst_type_str == "hifloat8") or \ | 101 | + if (x.dtype.name == "bfloat16" and dst_type_str == "hifloat8") or ( |
| 89 | - (x.dtype.name == "hifloat8" and dst_type_str == "bfloat16"): | 102 | + x.dtype.name == "hifloat8" and dst_type_str == "bfloat16" |
| 103 | + ): | ||
| 90 | np_dtype = _resolve_custom_numpy_dtype(dst_type_str) | 104 | np_dtype = _resolve_custom_numpy_dtype(dst_type_str) |
| 91 | return x.astype(np.float32).astype(np_dtype) | 105 | return x.astype(np.float32).astype(np_dtype) |
| 92 | elif dst_type_str in _SPECIAL_DTYPES: | 106 | elif dst_type_str in _SPECIAL_DTYPES: |
| @@ -102,3 +116,15 @@ def cast_golden(x, | |||
| 102 | return x.astype(np.bool_) | 116 | return x.astype(np.bool_) |
| 103 | else: | 117 | else: |
| 104 | return x.astype(getattr(np, dst_type_str)) | 118 | return x.astype(getattr(np, dst_type_str)) |
| 119 | + | ||
| 120 | + | ||
| 121 | +def aclnn_cast_golden(self, dtype=0, out=None, **kwargs): | ||
| 122 | + """ | ||
| 123 | + Aclnn golden for aclnnCast. | ||
| 124 | + Parameters follow @aclnnCastGetWorkspaceSize without workspaceSize & executor. | ||
| 125 | + All the input Tensors are torch.Tensor. | ||
| 126 | + """ | ||
| 127 | + from ttk.utilities import acl_to_torch_dtype | ||
| 128 | + | ||
| 129 | + torch_dtype = acl_to_torch_dtype([dtype])[0] | ||
| 130 | + return self.to(dtype=torch_dtype) | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,attributes,absolute_precision | ||
| 2 | +cast_fuzz_069,aclnnCast,"('uint8','float16')","('ND',)","((24, 16),(24, 16))","((0, 255),)","(-1,)",{'dtype': 1},0.0001 | ||
| 3 | +cast_fuzz_049,aclnnCast,"('int64','int32')","('ND',)","((3, 45),(3, 45))","((-9223372036854775808, 9223372036854775807),)","(-1,)",{'dtype': 3},0.0001 | ||
| 4 | +cast_fuzz_059,aclnnCast,"('int16','int8')","('ND',)","((1,), (1,))","((-32768, 32767),)","(-1,)",{'dtype': 2},0.001 | ||
| 5 | +cast_fuzz_050,aclnnCast,"('int64','int16')","('ND',)","((25, 1, 32),(25, 1, 32))","((-9223372036854775808, 9223372036854775807),)","(-1,)",{'dtype': 6},0.001 | ||
| 6 | +cast_fuzz_003,aclnnCast,"('bfloat16','float32')","('ND',)","((14, 17),(14, 17))","((-3.389531389251505e+38, 3.389531389251505e+38),)","(-1,)",{'dtype': 0},0.0001 | ||
| @@ -9,23 +9,34 @@ | |||
| 9 | # INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. | 9 | # INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. |
| 10 | # See LICENSE in the root of the software repository for the full text of the License. | 10 | # See LICENSE in the root of the software repository for the full text of the License. |
| 11 | # ---------------------------------------------------------------------------- | 11 | # ---------------------------------------------------------------------------- |
| 12 | - | 12 | + |
| 13 | import numpy as np | 13 | import numpy as np |
| 14 | - | 14 | +import torch |
| 15 | + | ||
| 15 | __golden__ = { | 16 | __golden__ = { |
| 16 | - "kernel": { | 17 | + "aclnn": { |
| 17 | - "ceil": "ceil_golden" | 18 | + "aclnnCeil": "aclnn_ceil_golden", |
| 18 | - } | 19 | + }, |
| 20 | + "kernel": {"ceil": "ceil_golden"}, | ||
| 19 | } | 21 | } |
| 20 | - | 22 | + |
| 21 | -def ceil_golden(x, | 23 | + |
| 22 | - **kwargs): | 24 | +def ceil_golden(x, **kwargs): |
| 23 | - ''' | 25 | + """ |
| 24 | Kernel golden for ceil. | 26 | Kernel golden for ceil. |
| 25 | All the parameters follow @ceil_def.cpp without outputs. | 27 | All the parameters follow @ceil_def.cpp without outputs. |
| 26 | All the input Tensors are numpy.ndarray. | 28 | All the input Tensors are numpy.ndarray. |
| 27 | kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, | 29 | kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, |
| 28 | - input_formats, output_formats, input_ori_formats, output_ori_formats, | 30 | + input_formats, output_formats, input_ori_formats, output_ori_formats, |
| 29 | - input_dtypes, output_dtypes. | 31 | + input_dtypes, output_dtypes. |
| 30 | - ''' | 32 | + """ |
| 31 | return np.ceil(x) | 33 | return np.ceil(x) |
| 34 | + | ||
| 35 | + | ||
| 36 | +def aclnn_ceil_golden(self, out=None, **kwargs): | ||
| 37 | + """ | ||
| 38 | + Aclnn golden for aclnnCeil. | ||
| 39 | + Parameters follow @aclnnCeilGetWorkspaceSize without workspaceSize & executor. | ||
| 40 | + All the input Tensors are torch.Tensor. | ||
| 41 | + """ | ||
| 42 | + return [torch.ceil(self)] | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,absolute_precision | ||
| 2 | +Ceil_float32_ND_fuzz_1,aclnnCeil,"('float32', 'float32')","('ND',)","((196, 2, 1, 76),(196, 2, 1, 76))","((-1000, -10),)","(-1,)","('float32',)",0.0001 | ||
| 3 | +Ceil_float16_ND_fuzz_2,aclnnCeil,"('float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('float16',)",0.001 | ||
| 4 | +Ceil_float32_ND_fuzz_3,aclnnCeil,"('float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)","('float32',)",0.0001 | ||
| 5 | +Ceil_float16_ND_fuzz_4,aclnnCeil,"('float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('float16',)",0.001 | ||
| 6 | +Ceil_float16_ND_fuzz_5,aclnnCeil,"('float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)","('float16',)",0.001 | ||
| @@ -1,4 +1,4 @@ | |||
| 1 | - #!/usr/bin/env python3 | 1 | +#!/usr/bin/env python3 |
| 2 | # -*- coding: UTF-8 -*- | 2 | # -*- coding: UTF-8 -*- |
| 3 | # ---------------------------------------------------------------------------- | 3 | # ---------------------------------------------------------------------------- |
| 4 | # Copyright (c) 2026 Huawei Technologies Co., Ltd. | 4 | # Copyright (c) 2026 Huawei Technologies Co., Ltd. |
| @@ -11,27 +11,38 @@ | |||
| 11 | # ---------------------------------------------------------------------------- | 11 | # ---------------------------------------------------------------------------- |
| 12 | 12 | ||
| 13 | import numpy as np | 13 | import numpy as np |
| 14 | +import torch | ||
| 14 | 15 | ||
| 15 | 16 | ||
| 16 | __golden__ = { | 17 | __golden__ = { |
| 17 | - "kernel": { | 18 | + "aclnn": { |
| 18 | - "cos": "cos_golden" | 19 | + "aclnnCos": "aclnn_cos_golden", |
| 19 | - } | 20 | + }, |
| 21 | + "kernel": {"cos": "cos_golden"}, | ||
| 20 | } | 22 | } |
| 21 | 23 | ||
| 22 | 24 | ||
| 23 | def cos_golden(x, **kwargs): | 25 | def cos_golden(x, **kwargs): |
| 24 | - ''' | 26 | + """ |
| 25 | Kernel golden for cos. | 27 | Kernel golden for cos. |
| 26 | All the parameters follow @cos_def.cpp without outputs. | 28 | All the parameters follow @cos_def.cpp without outputs. |
| 27 | All the input Tensors are numpy.ndarray. | 29 | All the input Tensors are numpy.ndarray. |
| 28 | kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, | 30 | kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, |
| 29 | input_formats, output_formats, input_ori_formats, output_ori_formats, | 31 | input_formats, output_formats, input_ori_formats, output_ori_formats, |
| 30 | input_dtypes, output_dtypes. | 32 | input_dtypes, output_dtypes. |
| 31 | - ''' | 33 | + """ |
| 32 | ori_dtype = x.dtype | 34 | ori_dtype = x.dtype |
| 33 | if ori_dtype.name in ("float16", "bfloat16"): | 35 | if ori_dtype.name in ("float16", "bfloat16"): |
| 34 | x_cast = x.astype(np.float32) | 36 | x_cast = x.astype(np.float32) |
| 35 | res = np.cos(x_cast) | 37 | res = np.cos(x_cast) |
| 36 | return res.astype(ori_dtype, copy=False) | 38 | return res.astype(ori_dtype, copy=False) |
| 37 | - return np.cos(x) | 39 | + return np.cos(x) |
| 40 | + | ||
| 41 | + | ||
| 42 | +def aclnn_cos_golden(input, out=None, **kwargs): | ||
| 43 | + """ | ||
| 44 | + Aclnn golden for aclnnCos. | ||
| 45 | + Parameters follow @aclnnCosGetWorkspaceSize without workspaceSize & executor. | ||
| 46 | + All the input Tensors are torch.Tensor. | ||
| 47 | + """ | ||
| 48 | + return [torch.cos(input)] | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,absolute_precision | ||
| 2 | +Cos_float32_ND_fuzz_1,aclnnCos,"('float32', 'float32')","('ND',)","((196, 2, 1, 76),(196, 2, 1, 76))","((-1000, -10),)","(-1,)","('float32',)",0.0001 | ||
| 3 | +Cos_float16_ND_fuzz_2,aclnnCos,"('float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('float16',)",0.001 | ||
| 4 | +Cos_float32_ND_fuzz_3,aclnnCos,"('float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)","('float32',)",0.0001 | ||
| 5 | +Cos_float16_ND_fuzz_4,aclnnCos,"('float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('float16',)",0.001 | ||
| 6 | +Cos_float16_ND_fuzz_5,aclnnCos,"('float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)","('float16',)",0.001 | ||
| @@ -14,26 +14,27 @@ import numpy as np | |||
| 14 | 14 | ||
| 15 | 15 | ||
| 16 | __golden__ = { | 16 | __golden__ = { |
| 17 | - "kernel": { | 17 | + "aclnn": { |
| 18 | - "erf": "erf_golden" | 18 | + "aclnnErf": "aclnn_erf_golden", |
| 19 | - } | 19 | + }, |
| 20 | + "kernel": {"erf": "erf_golden"}, | ||
| 20 | } | 21 | } |
| 21 | 22 | ||
| 22 | 23 | ||
| 23 | def erf_golden(x, **kwargs): | 24 | def erf_golden(x, **kwargs): |
| 24 | - ''' | 25 | + """ |
| 25 | Kernel golden for erf. | 26 | Kernel golden for erf. |
| 26 | All the parameters follow @erf_def.cpp without outputs. | 27 | All the parameters follow @erf_def.cpp without outputs. |
| 27 | All the input Tensors are numpy.ndarray. | 28 | All the input Tensors are numpy.ndarray. |
| 28 | kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, | 29 | kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, |
| 29 | input_formats, output_formats, input_ori_formats, output_ori_formats, | 30 | input_formats, output_formats, input_ori_formats, output_ori_formats, |
| 30 | input_dtypes, output_dtypes. | 31 | input_dtypes, output_dtypes. |
| 31 | - ''' | 32 | + """ |
| 32 | import torch | 33 | import torch |
| 33 | - | 34 | + |
| 34 | ori_dtype = kwargs.get("input_dtypes", ["float32"])[0] | 35 | ori_dtype = kwargs.get("input_dtypes", ["float32"])[0] |
| 35 | x_dtype = x.dtype | 36 | x_dtype = x.dtype |
| 36 | - | 37 | + |
| 37 | if ori_dtype and "bfloat16" in str(ori_dtype).lower(): | 38 | if ori_dtype and "bfloat16" in str(ori_dtype).lower(): |
| 38 | x_tensor = torch.from_numpy(x.astype(np.float32)) | 39 | x_tensor = torch.from_numpy(x.astype(np.float32)) |
| 39 | output = torch.erf(x_tensor) | 40 | output = torch.erf(x_tensor) |
| @@ -45,4 +46,15 @@ def erf_golden(x, **kwargs): | |||
| 45 | else: | 46 | else: |
| 46 | x_tensor = torch.from_numpy(x) | 47 | x_tensor = torch.from_numpy(x) |
| 47 | output = torch.erf(x_tensor) | 48 | output = torch.erf(x_tensor) |
| 48 | - return output.numpy() | 49 | + return output.numpy() |
| 50 | + | ||
| 51 | + | ||
| 52 | +def aclnn_erf_golden(self, out=None, **kwargs): | ||
| 53 | + """ | ||
| 54 | + Aclnn golden for aclnnErf. | ||
| 55 | + Parameters follow @aclnnErfGetWorkspaceSize without workspaceSize & executor. | ||
| 56 | + All the input Tensors are torch.Tensor. | ||
| 57 | + """ | ||
| 58 | + import torch | ||
| 59 | + | ||
| 60 | + return [torch.erf(self)] | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes | ||
| 2 | +Erf_float32_ND_fuzz_1,aclnnErf,"('float32', 'float32')","('ND',)","((196, 2, 1, 76),(196, 2, 1, 76))","((-1000, -10),)","(-1,)","('float32',)" | ||
| 3 | +Erf_float16_ND_fuzz_2,aclnnErf,"('float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('float16',)" | ||
| 4 | +Erf_float32_ND_fuzz_3,aclnnErf,"('float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)","('float32',)" | ||
| 5 | +Erf_float16_ND_fuzz_4,aclnnErf,"('float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('float16',)" | ||
| 6 | +Erf_bfloat16_ND_fuzz_5,aclnnErf,"('bfloat16', 'bfloat16')","('ND',)","((1, 64, 10, 1, 192, 1), (1, 64, 10, 1, 192, 1))","((-1000, -10),)","(-1,)","('bfloat16',)" | ||
| @@ -12,22 +12,22 @@ | |||
| 12 | import numpy as np | 12 | import numpy as np |
| 13 | 13 | ||
| 14 | __golden__ = { | 14 | __golden__ = { |
| 15 | - "kernel": { | 15 | + "aclnn": { |
| 16 | - "exp": "exp_golden" | 16 | + "aclnnExp": "aclnn_exp_golden", |
| 17 | - } | 17 | + }, |
| 18 | + "kernel": {"exp": "exp_golden"}, | ||
| 18 | } | 19 | } |
| 19 | - | 20 | + |
| 20 | -def exp_golden(x, | 21 | + |
| 21 | - base: float=-1.0, scale: float=1.0, shift: float=0.0, | 22 | +def exp_golden(x, base: float = -1.0, scale: float = 1.0, shift: float = 0.0, **kwargs): |
| 22 | - **kwargs): | 23 | + """ |
| 23 | - ''' | ||
| 24 | Kernel golden for exp. | 24 | Kernel golden for exp. |
| 25 | All the parameters follow @exp_def.cpp without outputs. | 25 | All the parameters follow @exp_def.cpp without outputs. |
| 26 | All the input Tensors are numpy.ndarray. | 26 | All the input Tensors are numpy.ndarray. |
| 27 | - kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, | 27 | + kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, |
| 28 | - input_formats, output_formats, input_ori_formats, output_ori_formats, | 28 | + input_formats, output_formats, input_ori_formats, output_ori_formats, |
| 29 | - input_dtypes, output_dtypes. | 29 | + input_dtypes, output_dtypes. |
| 30 | - ''' | 30 | + """ |
| 31 | import torch | 31 | import torch |
| 32 | 32 | ||
| 33 | x_dtype = x.dtype | 33 | x_dtype = x.dtype |
| @@ -39,10 +39,21 @@ def exp_golden(x, | |||
| 39 | if base == -1: | 39 | if base == -1: |
| 40 | output = torch.exp(x) | 40 | output = torch.exp(x) |
| 41 | else: | 41 | else: |
| 42 | - output = torch.exp((scale * x + shift)*np.log(base)) | 42 | + output = torch.exp((scale * x + shift) * np.log(base)) |
| 43 | elif base == -1: | 43 | elif base == -1: |
| 44 | output = torch.exp(scale * x + shift) | 44 | output = torch.exp(scale * x + shift) |
| 45 | else: | 45 | else: |
| 46 | - output = torch.exp((scale * x + shift)*np.log(base)) | 46 | + output = torch.exp((scale * x + shift) * np.log(base)) |
| 47 | 47 | ||
| 48 | return output.numpy().astype(x_dtype, copy=False) | 48 | return output.numpy().astype(x_dtype, copy=False) |
| 49 | + | ||
| 50 | + | ||
| 51 | +def aclnn_exp_golden(self, out=None, **kwargs): | ||
| 52 | + """ | ||
| 53 | + Aclnn golden for aclnnExp. | ||
| 54 | + Parameters follow @aclnnExpGetWorkspaceSize without workspaceSize & executor. | ||
| 55 | + All the input Tensors are torch.Tensor. | ||
| 56 | + """ | ||
| 57 | + import torch | ||
| 58 | + | ||
| 59 | + return [torch.exp(self)] | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,absolute_precision | ||
| 2 | +Exp_float32_ND_fuzz_1,aclnnExp,"('float32', 'float32')","('ND',)","((196, 2, 1, 76),(196, 2, 1, 76))","((-1000, -10),)","(-1,)","('float32',)",0.0001 | ||
| 3 | +Exp_float16_ND_fuzz_2,aclnnExp,"('float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('float16',)",0.001 | ||
| 4 | +Exp_float32_ND_fuzz_3,aclnnExp,"('float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)","('float32',)",0.0001 | ||
| 5 | +Exp_float16_ND_fuzz_4,aclnnExp,"('float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('float16',)",0.001 | ||
| 6 | +Exp_float16_ND_fuzz_5,aclnnExp,"('float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)","('float16',)",0.001 | ||
| @@ -11,14 +11,25 @@ | |||
| 11 | # ---------------------------------------------------------------------------- | 11 | # ---------------------------------------------------------------------------- |
| 12 | 12 | ||
| 13 | import numpy as np | 13 | import numpy as np |
| 14 | +import torch | ||
| 14 | 15 | ||
| 15 | 16 | ||
| 16 | __golden__ = { | 17 | __golden__ = { |
| 17 | - "kernel": { | 18 | + "aclnn": { |
| 18 | - "is_finite": "is_finite_golden" | 19 | + "aclnnIsFinite": "aclnn_is_finite_golden", |
| 19 | - } | 20 | + }, |
| 21 | + "kernel": {"is_finite": "is_finite_golden"}, | ||
| 20 | } | 22 | } |
| 21 | 23 | ||
| 22 | 24 | ||
| 23 | def is_finite_golden(x, **kwargs): | 25 | def is_finite_golden(x, **kwargs): |
| 24 | - return np.isfinite(x) | 26 | + return np.isfinite(x) |
| 27 | + | ||
| 28 | + | ||
| 29 | +def aclnn_is_finite_golden(self, out=None, **kwargs): | ||
| 30 | + """ | ||
| 31 | + Aclnn golden for aclnnIsFinite. | ||
| 32 | + Parameters follow @aclnnIsFiniteGetWorkspaceSize without workspaceSize & executor. | ||
| 33 | + All the input Tensors are torch.Tensor. | ||
| 34 | + """ | ||
| 35 | + return [torch.ops.aten.isfinite(self)] | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +,testcase_name,api_name,tensor_view_shapes,tensor_dtypes | ||
| 2 | +0,aclnnIsFinite_00000,aclnnIsFinite,"[[38],[38]]","('float16','bool')" | ||
| 3 | +1,aclnnIsFinite_00001,aclnnIsFinite,"[[47, 15, 8, 21, 5],[47, 15, 8, 21, 5]]","('float32','bool')" | ||
| 4 | +2,aclnnIsFinite_00002,aclnnIsFinite,"[[10, 42, 4, 15],[10, 42, 4, 15]]","('float16','bool')" | ||
| 5 | +3,aclnnIsFinite_00003,aclnnIsFinite,"[[43, 22],[43, 22]]","('float16','bool')" | ||
| 6 | +4,aclnnIsFinite_00004,aclnnIsFinite,"[[9, 14, 29],[9, 14, 29]]","('bfloat16','bool')" | ||
| @@ -11,14 +11,20 @@ | |||
| 11 | # ---------------------------------------------------------------------------- | 11 | # ---------------------------------------------------------------------------- |
| 12 | 12 | ||
| 13 | import numpy as np | 13 | import numpy as np |
| 14 | +import torch | ||
| 14 | 15 | ||
| 15 | 16 | ||
| 16 | __golden__ = { | 17 | __golden__ = { |
| 17 | - "kernel": { | 18 | + "aclnn": { |
| 18 | - "is_inf": "is_inf_golden" | 19 | + "aclnnIsInf": "aclnn_is_inf_golden", |
| 19 | - } | 20 | + }, |
| 21 | + "kernel": {"is_inf": "is_inf_golden"}, | ||
| 20 | } | 22 | } |
| 21 | 23 | ||
| 22 | 24 | ||
| 23 | def is_inf_golden(x, **kwargs): | 25 | def is_inf_golden(x, **kwargs): |
| 24 | - return np.isinf(x) | 26 | + return np.isinf(x) |
| 27 | + | ||
| 28 | + | ||
| 29 | +def aclnn_is_inf_golden(x, out=None, **kwargs): | ||
| 30 | + return [torch.isinf(x)] | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes | ||
| 2 | +IsInf_float32_ND_fuzz_1,aclnnIsInf,"('float32', 'bool')","('ND',)","((196, 2, 1, 76),(196, 2, 1, 76))","((-1000, -10),)","(-1,)","('bool',)" | ||
| 3 | +IsInf_float16_ND_fuzz_2,aclnnIsInf,"('float16', 'bool')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('bool',)" | ||
| 4 | +IsInf_float32_ND_fuzz_3,aclnnIsInf,"('float32', 'bool')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)","('bool',)" | ||
| 5 | +IsInf_float16_ND_fuzz_4,aclnnIsInf,"('float16', 'bool')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('bool',)" | ||
| 6 | +IsInf_float16_ND_fuzz_5,aclnnIsInf,"('float16', 'bool')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((inf, inf),)","(-1,)","('bool',)" | ||
| @@ -13,28 +13,28 @@ import numpy as np | |||
| 13 | import torch | 13 | import torch |
| 14 | 14 | ||
| 15 | __golden__ = { | 15 | __golden__ = { |
| 16 | - "kernel": { | 16 | + "aclnn": { |
| 17 | - "log": "log_golden" | 17 | + "aclnnLog": "aclnn_log_golden", |
| 18 | - } | 18 | + }, |
| 19 | + "kernel": {"log": "log_golden"}, | ||
| 19 | } | 20 | } |
| 20 | - | 21 | + |
| 21 | -def log_golden(x, | 22 | + |
| 22 | - base: float=-1.0, scale: float=1.0, shift: float=0.0, | 23 | +def log_golden(x, base: float = -1.0, scale: float = 1.0, shift: float = 0.0, **kwargs): |
| 23 | - **kwargs): | 24 | + """ |
| 24 | - ''' | ||
| 25 | Kernel golden for log. | 25 | Kernel golden for log. |
| 26 | All the parameters follow @log_def.cpp without outputs. | 26 | All the parameters follow @log_def.cpp without outputs. |
| 27 | All the input Tensors are numpy.ndarray. | 27 | All the input Tensors are numpy.ndarray. |
| 28 | - kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, | 28 | + kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, |
| 29 | - input_formats, output_formats, input_ori_formats, output_ori_formats, | 29 | + input_formats, output_formats, input_ori_formats, output_ori_formats, |
| 30 | - input_dtypes, output_dtypes. | 30 | + input_dtypes, output_dtypes. |
| 31 | - ''' | 31 | + """ |
| 32 | x_dtype = x.dtype | 32 | x_dtype = x.dtype |
| 33 | if x_dtype.name == "bfloat16" or x_dtype.name == "float16": | 33 | if x_dtype.name == "bfloat16" or x_dtype.name == "float16": |
| 34 | x = torch.from_numpy(x.astype(np.float32)) | 34 | x = torch.from_numpy(x.astype(np.float32)) |
| 35 | - else : | 35 | + else: |
| 36 | x = torch.from_numpy(x) | 36 | x = torch.from_numpy(x) |
| 37 | - | 37 | + |
| 38 | if scale == 1 and shift == 0: | 38 | if scale == 1 and shift == 0: |
| 39 | if base == -1: | 39 | if base == -1: |
| 40 | output = torch.log(x) | 40 | output = torch.log(x) |
| @@ -43,10 +43,21 @@ def log_golden(x, | |||
| 43 | elif base == 10: | 43 | elif base == 10: |
| 44 | output = torch.log10(x) | 44 | output = torch.log10(x) |
| 45 | else: | 45 | else: |
| 46 | - output = torch.log((scale * x + shift))/np.log(base) | 46 | + output = torch.log((scale * x + shift)) / np.log(base) |
| 47 | elif base == -1: | 47 | elif base == -1: |
| 48 | output = torch.log((scale * x + shift)) | 48 | output = torch.log((scale * x + shift)) |
| 49 | else: | 49 | else: |
| 50 | - output = torch.log((scale * x + shift))/np.log(base) | 50 | + output = torch.log((scale * x + shift)) / np.log(base) |
| 51 | 51 | ||
| 52 | return output.numpy().astype(x_dtype, copy=False) | 52 | return output.numpy().astype(x_dtype, copy=False) |
| 53 | + | ||
| 54 | + | ||
| 55 | +def aclnn_log_golden(self, out=None, **kwargs): | ||
| 56 | + """ | ||
| 57 | + Aclnn golden for aclnnLog. | ||
| 58 | + Parameters follow @aclnnLogGetWorkspaceSize without workspaceSize & executor. | ||
| 59 | + All the input Tensors are torch.Tensor. | ||
| 60 | + """ | ||
| 61 | + import torch | ||
| 62 | + | ||
| 63 | + return [torch.log(self)] | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,absolute_precision | ||
| 2 | +Log_float32_ND_fuzz_1,aclnnLog,"('float32', 'float32')","('ND',)","((196, 2, 1, 76),(196, 2, 1, 76))","((-1000, -10),)","(-1,)","('float32',)",0.0001 | ||
| 3 | +Log_float16_ND_fuzz_2,aclnnLog,"('float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('float16',)",0.001 | ||
| 4 | +Log_float32_ND_fuzz_3,aclnnLog,"('float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)","('float32',)",0.0001 | ||
| 5 | +Log_float16_ND_fuzz_4,aclnnLog,"('float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('float16',)",0.001 | ||
| 6 | +Log_float16_ND_fuzz_5,aclnnLog,"('float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)","('float16',)",0.001 | ||
| @@ -15,19 +15,31 @@ import torch | |||
| 15 | 15 | ||
| 16 | 16 | ||
| 17 | __golden__ = { | 17 | __golden__ = { |
| 18 | - "kernel": { | 18 | + "aclnn": { |
| 19 | - "log1p": "log1p_golden" | 19 | + "aclnnLog1p": "aclnn_log1p_golden", |
| 20 | - } | 20 | + }, |
| 21 | + "kernel": {"log1p": "log1p_golden"}, | ||
| 21 | } | 22 | } |
| 22 | 23 | ||
| 23 | 24 | ||
| 24 | def log1p_golden(x, **kwargs): | 25 | def log1p_golden(x, **kwargs): |
| 25 | ori_dtype = kwargs.get("input_dtypes", ["float32"])[0] | 26 | ori_dtype = kwargs.get("input_dtypes", ["float32"])[0] |
| 26 | x_dtype = x.dtype | 27 | x_dtype = x.dtype |
| 27 | - | 28 | + |
| 28 | if "bfloat16" in str(ori_dtype).lower() or "float16" in str(ori_dtype).lower(): | 29 | if "bfloat16" in str(ori_dtype).lower() or "float16" in str(ori_dtype).lower(): |
| 29 | x_tensor = torch.from_numpy(x.astype(np.float32)) | 30 | x_tensor = torch.from_numpy(x.astype(np.float32)) |
| 30 | output = torch.log1p(x_tensor) | 31 | output = torch.log1p(x_tensor) |
| 31 | return output.numpy().astype(x_dtype, copy=False) | 32 | return output.numpy().astype(x_dtype, copy=False) |
| 32 | else: | 33 | else: |
| 33 | - return np.log1p(x) | 34 | + return np.log1p(x) |
| 35 | + | ||
| 36 | + | ||
| 37 | +def aclnn_log1p_golden(self, out=None, **kwargs): | ||
| 38 | + """ | ||
| 39 | + Aclnn golden for aclnnLog1p. | ||
| 40 | + Parameters follow @aclnnLog1pGetWorkspaceSize without workspaceSize & executor. | ||
| 41 | + All the input Tensors are torch.Tensor. | ||
| 42 | + """ | ||
| 43 | + import torch | ||
| 44 | + | ||
| 45 | + return [torch.log1p(self)] | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,absolute_precision | ||
| 2 | +Log1p_float32_ND_fuzz_1,aclnnLog1p,"('float32', 'float32')","('ND',)","((196, 2, 1, 76),(196, 2, 1, 76))","((-1000, -10),)","(-1,)","('float32',)",0.0001 | ||
| 3 | +Log1p_float16_ND_fuzz_2,aclnnLog1p,"('float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('float16',)",0.001 | ||
| 4 | +Log1p_float32_ND_fuzz_3,aclnnLog1p,"('float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+30, 3.4e+30),)","(-1,)","('float32',)",0.0001 | ||
| 5 | +Log1p_float16_ND_fuzz_4,aclnnLog1p,"('float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('float16',)",0.001 | ||
| 6 | +Log1p_float16_ND_fuzz_5,aclnnLog1p,"('float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)","('float16',)",0.001 | ||
| @@ -11,14 +11,20 @@ | |||
| 11 | # ---------------------------------------------------------------------------- | 11 | # ---------------------------------------------------------------------------- |
| 12 | 12 | ||
| 13 | import numpy as np | 13 | import numpy as np |
| 14 | +import torch | ||
| 14 | 15 | ||
| 15 | 16 | ||
| 16 | __golden__ = { | 17 | __golden__ = { |
| 17 | - "kernel": { | 18 | + "aclnn": { |
| 18 | - "logical_not": "logical_not_golden" | 19 | + "aclnnLogicalNot": "aclnn_logical_not_golden", |
| 19 | - } | 20 | + }, |
| 21 | + "kernel": {"logical_not": "logical_not_golden"}, | ||
| 20 | } | 22 | } |
| 21 | 23 | ||
| 22 | 24 | ||
| 23 | def logical_not_golden(x, **kwargs): | 25 | def logical_not_golden(x, **kwargs): |
| 24 | - return np.logical_not(x) | 26 | + return np.logical_not(x) |
| 27 | + | ||
| 28 | + | ||
| 29 | +def aclnn_logical_not_golden(self, out=None, **kwargs): | ||
| 30 | + return [torch.logical_not(self)] | ||
| @@ -0,0 +1,2 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,absolute_precision | ||
| 2 | +LogicalNot_float32_ND_fuzz_1,aclnnLogicalNot,"('float32',)","('ND',)","((196, 2, 1, 76),(196, 2, 1, 76))","((-2, 2),)","(-1,)","('float32',)",0.0001 | ||
| @@ -9,30 +9,42 @@ | |||
| 9 | # INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. | 9 | # INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. |
| 10 | # See LICENSE in the root of the software repository for the full text of the License. | 10 | # See LICENSE in the root of the software repository for the full text of the License. |
| 11 | # ---------------------------------------------------------------------------- | 11 | # ---------------------------------------------------------------------------- |
| 12 | -import numpy as np | 12 | +import torch |
| 13 | 13 | ||
| 14 | __golden__ = { | 14 | __golden__ = { |
| 15 | - "kernel": { | 15 | + "aclnn": { |
| 16 | - "masked_scale": "masked_scale_golden" | 16 | + "aclnnMaskedScale": "aclnn_masked_scale_golden", |
| 17 | - } | 17 | + }, |
| 18 | + "kernel": {"masked_scale": "masked_scale_golden"}, | ||
| 18 | } | 19 | } |
| 19 | - | 20 | + |
| 20 | -def masked_scale_golden(x, mask, value: float, | 21 | + |
| 21 | - **kwargs): | 22 | +def masked_scale_golden(x, mask, value: float, **kwargs): |
| 22 | - ''' | 23 | + """ |
| 23 | Kernel golden for masked_scale. | 24 | Kernel golden for masked_scale. |
| 24 | All the parameters follow @masked_scale_def.cpp without outputs. | 25 | All the parameters follow @masked_scale_def.cpp without outputs. |
| 25 | All the input Tensors are numpy.ndarray. | 26 | All the input Tensors are numpy.ndarray. |
| 26 | - kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, | 27 | + kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, |
| 27 | - input_formats, output_formats, input_ori_formats, output_ori_formats, | 28 | + input_formats, output_formats, input_ori_formats, output_ori_formats, |
| 28 | - input_dtypes, output_dtypes. | 29 | + input_dtypes, output_dtypes. |
| 29 | - ''' | 30 | + """ |
| 30 | x_dtype = x.dtype | 31 | x_dtype = x.dtype |
| 31 | 32 | ||
| 32 | - if x_dtype.name not in ('float32', 'float64'): | 33 | + if x_dtype.name not in ("float32", "float64"): |
| 33 | - x = x.astype('float32') | 34 | + x = x.astype("float32") |
| 34 | - if mask.dtype.name not in ('float32', 'float64'): | 35 | + if mask.dtype.name not in ("float32", "float64"): |
| 35 | - mask = mask.astype('float32') | 36 | + mask = mask.astype("float32") |
| 36 | res = x * mask * value | 37 | res = x * mask * value |
| 37 | 38 | ||
| 38 | return res.astype(x_dtype, copy=False) | 39 | return res.astype(x_dtype, copy=False) |
| 40 | + | ||
| 41 | + | ||
| 42 | +def aclnn_masked_scale_golden(self, mask, scale, out=None, **kwargs): | ||
| 43 | + if hasattr(scale, "item"): | ||
| 44 | + scale = scale.item() | ||
| 45 | + orig_dtype = self.dtype | ||
| 46 | + if orig_dtype in (torch.float16, torch.bfloat16): | ||
| 47 | + self = self.to(torch.float32) | ||
| 48 | + mask_f = mask.to(torch.float32) | ||
| 49 | + result = self * mask_f * scale | ||
| 50 | + return [result.to(orig_dtype)] | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,attributes,absolute_precision | ||
| 2 | +MaskedScale_ND_fuzz_1,aclnnMaskedScale,"('float32', 'float32', 'float32')","('ND',)","((196, 2, 1, 76, 1, 1), (196, 2, 1, 76, 1, 1), (196, 2, 1, 76, 1, 1))","((-1000, -10),)","(-1,)",{'scale':0.12},0.0001 | ||
| 3 | +MaskedScale_ND_fuzz_2,aclnnMaskedScale,"('float16', 'float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)",{'scale':0.12},0.001 | ||
| 4 | +MaskedScale_ND_fuzz_3,aclnnMaskedScale,"('float32', 'float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)",{'scale':0.12},0.0001 | ||
| 5 | +MaskedScale_ND_fuzz_4,aclnnMaskedScale,"('float16', 'float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)",{'scale':0.12},0.001 | ||
| 6 | +MaskedScale_ND_fuzz_5,aclnnMaskedScale,"('float16', 'float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)",{'scale':0.12},0.001 | ||
| @@ -15,19 +15,31 @@ import torch | |||
| 15 | 15 | ||
| 16 | 16 | ||
| 17 | __golden__ = { | 17 | __golden__ = { |
| 18 | - "kernel": { | 18 | + "aclnn": { |
| 19 | - "neg": "neg_golden" | 19 | + "aclnnNeg": "aclnn_neg_golden", |
| 20 | - } | 20 | + }, |
| 21 | + "kernel": {"neg": "neg_golden"}, | ||
| 21 | } | 22 | } |
| 22 | 23 | ||
| 23 | 24 | ||
| 24 | def neg_golden(x, **kwargs): | 25 | def neg_golden(x, **kwargs): |
| 25 | ori_dtype = kwargs.get("input_dtypes", ["float32"])[0] | 26 | ori_dtype = kwargs.get("input_dtypes", ["float32"])[0] |
| 26 | x_dtype = x.dtype | 27 | x_dtype = x.dtype |
| 27 | - | 28 | + |
| 28 | if "bfloat16" in str(ori_dtype).lower() or "float16" in str(ori_dtype).lower(): | 29 | if "bfloat16" in str(ori_dtype).lower() or "float16" in str(ori_dtype).lower(): |
| 29 | x_tensor = torch.from_numpy(x.astype(np.float32)) | 30 | x_tensor = torch.from_numpy(x.astype(np.float32)) |
| 30 | output = torch.neg(x_tensor) | 31 | output = torch.neg(x_tensor) |
| 31 | return output.numpy().astype(x_dtype, copy=False) | 32 | return output.numpy().astype(x_dtype, copy=False) |
| 32 | else: | 33 | else: |
| 33 | - return np.negative(x) | 34 | + return np.negative(x) |
| 35 | + | ||
| 36 | + | ||
| 37 | +def aclnn_neg_golden(self, out=None, **kwargs): | ||
| 38 | + """ | ||
| 39 | + Aclnn golden for aclnnNeg. | ||
| 40 | + Parameters follow @aclnnNegGetWorkspaceSize without workspaceSize & executor. | ||
| 41 | + All the input Tensors are torch.Tensor. | ||
| 42 | + """ | ||
| 43 | + import torch | ||
| 44 | + | ||
| 45 | + return [torch.neg(self)] | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,absolute_precision | ||
| 2 | +Neg_float32_ND_fuzz_1,aclnnNeg,"('float32', 'float32')","('ND',)","((196, 2, 1, 76, 1, 1), (196, 2, 1, 76, 1, 1))","((-1000, -10),)","(-1,)",0.0001 | ||
| 3 | +Neg_float16_ND_fuzz_2,aclnnNeg,"('float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)",0.001 | ||
| 4 | +Neg_float32_ND_fuzz_3,aclnnNeg,"('float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)",0.0001 | ||
| 5 | +Neg_float16_ND_fuzz_4,aclnnNeg,"('float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)",0.001 | ||
| 6 | +Neg_float16_ND_fuzz_5,aclnnNeg,"('float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)",0.001 | ||
| @@ -11,7 +11,14 @@ | |||
| 11 | # ---------------------------------------------------------------------------- | 11 | # ---------------------------------------------------------------------------- |
| 12 | 12 | ||
| 13 | 13 | ||
| 14 | -__golden__ = {"kernel": {"reduce_max": "reduce_max_golden"}} | 14 | +__golden__ = { |
| 15 | + "aclnn": { | ||
| 16 | + "aclnnAmax": "aclnn_amax_golden", | ||
| 17 | + "aclnnMax": "aclnn_max_golden", | ||
| 18 | + "aclnnMaxV2": "aclnn_max_v2_golden", | ||
| 19 | + }, | ||
| 20 | + "kernel": {"reduce_max": "reduce_max_golden"}, | ||
| 21 | +} | ||
| 15 | 22 | ||
| 16 | 23 | ||
| 17 | def reduce_max_golden(x, axes=None, keep_dims: bool = False, **kwargs): | 24 | def reduce_max_golden(x, axes=None, keep_dims: bool = False, **kwargs): |
| @@ -55,3 +62,51 @@ def reduce_max_golden(x, axes=None, keep_dims: bool = False, **kwargs): | |||
| 55 | # Fallback to NumPy | 62 | # Fallback to NumPy |
| 56 | res = np.max(x, axis=axis, keepdims=keep_dims) | 63 | res = np.max(x, axis=axis, keepdims=keep_dims) |
| 57 | return res.astype(input_dtype, copy=False) | 64 | return res.astype(input_dtype, copy=False) |
| 65 | + | ||
| 66 | + | ||
| 67 | +def aclnn_max_v2_golden( | ||
| 68 | + self, dims=0, keepDims=0, noopWithEmptyDims=0, out=None, **kwargs | ||
| 69 | +): | ||
| 70 | + """ | ||
| 71 | + Aclnn golden for aclnnMaxV2. | ||
| 72 | + Parameters follow @aclnnMaxV2GetWorkspaceSize without workspaceSize & executor. | ||
| 73 | + All the input Tensors are torch.Tensor. | ||
| 74 | + """ | ||
| 75 | + import torch | ||
| 76 | + | ||
| 77 | + ipt = self | ||
| 78 | + dim = kwargs.get("attributes", {})["dims"] | ||
| 79 | + keepdim = kwargs.get("attributes", {})["keepDims"] | ||
| 80 | + noop_with_empty_dims = kwargs.get("attributes", {})["noopWithEmptyDims"] | ||
| 81 | + if dim is None or (isinstance(dim, (tuple, list)) and len(dim) == 0): | ||
| 82 | + if noop_with_empty_dims: | ||
| 83 | + result = ipt | ||
| 84 | + else: | ||
| 85 | + result = ipt.flatten() | ||
| 86 | + if keepdim: | ||
| 87 | + result = result.reshape([1] * ipt.dim()) | ||
| 88 | + else: | ||
| 89 | + result = torch.amax(ipt, dim=dim, keepdim=keepdim) | ||
| 90 | + return result | ||
| 91 | + | ||
| 92 | + | ||
| 93 | +def aclnn_max_golden(self, out=None, **kwargs): | ||
| 94 | + """ | ||
| 95 | + Aclnn golden for aclnnMax. | ||
| 96 | + Parameters follow @aclnnMaxGetWorkspaceSize without workspaceSize & executor. | ||
| 97 | + All the input Tensors are torch.Tensor. | ||
| 98 | + """ | ||
| 99 | + import torch | ||
| 100 | + | ||
| 101 | + return [torch.ops.aten.amax(self)] | ||
| 102 | + | ||
| 103 | + | ||
| 104 | +def aclnn_amax_golden(self, dim=0, keepDim=0, out=None, **kwargs): | ||
| 105 | + """ | ||
| 106 | + Aclnn golden for aclnnAmax. | ||
| 107 | + Parameters follow @aclnnAmaxGetWorkspaceSize without workspaceSize & executor. | ||
| 108 | + All the input Tensors are torch.Tensor. | ||
| 109 | + """ | ||
| 110 | + import torch | ||
| 111 | + | ||
| 112 | + return torch.amax(self, dim=dim, keepdim=keepDim) | ||
| @@ -0,0 +1,2 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,attributes,tensor_view_shapes,output_tensor_indexes,input_data_ranges,precision_tolerances,absolute_precision | ||
| 2 | +aclnnAmax_00,aclnnAmax,"('bfloat16', 'bfloat16')","{'dim': [-5, 4], 'keepDim': True}","((1,7831,1,1,107),(1,7831,1,1,1))","(1,)","[[0, 0.001]]","((0.001, 0.001),)",0.000001 | ||
| @@ -0,0 +1,2 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,attributes,tensor_view_shapes,output_tensor_indexes,input_data_ranges,precision_tolerances,absolute_precision | ||
| 2 | +aclnnMax_00,aclnnMax,"('float32', 'float32')",,"((2,3,4,32),(1,))","(1,)","[[0, 0.001]]","((0.001, 0.001),)",0.000001 | ||
| @@ -11,47 +11,80 @@ | |||
| 11 | # ---------------------------------------------------------------------------- | 11 | # ---------------------------------------------------------------------------- |
| 12 | 12 | ||
| 13 | 13 | ||
| 14 | -__golden__ = {"kernel": {"reduce_min": "reduce_min_golden"}} | 14 | +import torch |
| 15 | 15 | ||
| 16 | 16 | ||
| 17 | -def reduce_min_golden(x, axes=None, keep_dims: bool = False, **kwargs): | 17 | +__golden__ = { |
| 18 | + "aclnn": { | ||
| 19 | + "aclnnAminmax": "aclnn_aminmax_golden", | ||
| 20 | + "aclnnAminmaxDim": "aclnn_aminmax_dim_golden", | ||
| 21 | + "aclnnAminmaxAll": "aclnn_aminmax_all_golden", | ||
| 22 | + }, | ||
| 23 | + "kernel": {"reduce_min": "reduce_min_golden"}, | ||
| 24 | +} | ||
| 25 | + | ||
| 26 | + | ||
| 27 | +# def reduce_min_golden(x, axes=None, keep_dims: bool = False, **kwargs): | ||
| 28 | +# """ | ||
| 29 | +# Kernel golden for reduce_min. | ||
| 30 | +# """ | ||
| 31 | +# import numpy as np | ||
| 32 | + | ||
| 33 | +# input_dtype = x.dtype | ||
| 34 | +# if str(input_dtype) == "bfloat16": | ||
| 35 | +# x = x.astype(np.float32) | ||
| 36 | + | ||
| 37 | +# if axes is not None: | ||
| 38 | +# axis = tuple(int(a) for a in np.asarray(axes).flatten()) | ||
| 39 | +# else: | ||
| 40 | +# axis = None | ||
| 41 | + | ||
| 42 | +# try: | ||
| 43 | +# import tensorflow as tf | ||
| 44 | +# x_tensor = tf.constant(x) | ||
| 45 | +# res_tensor = tf.reduce_min(x_tensor, axis=axis, keepdims=keep_dims) | ||
| 46 | +# res = res_tensor.numpy() | ||
| 47 | +# except ImportError: | ||
| 48 | +# try: | ||
| 49 | +# import torch | ||
| 50 | +# x_torch = torch.from_numpy(x) | ||
| 51 | +# if axis is not None: | ||
| 52 | +# res_torch = torch.amin(x_torch, dim=axis, keepdim=keep_dims) | ||
| 53 | +# else: | ||
| 54 | +# res_torch = torch.amin(x_torch, keepdim=keep_dims) | ||
| 55 | +# res = res_torch.numpy() | ||
| 56 | +# except ImportError: | ||
| 57 | +# res = np.min(x, axis=axis, keepdims=keep_dims) | ||
| 58 | +# return res.astype(input_dtype, copy=False) | ||
| 59 | + | ||
| 60 | + | ||
| 61 | +def aclnn_aminmax_golden(self, dim=0, keepDim=0, minOut=None, maxOut=None, **kwargs): | ||
| 18 | """ | 62 | """ |
| 19 | - Kernel golden for reduce_min. | 63 | + Aclnn golden for aclnnAminmax. |
| 20 | - All the parameters follow @reduce_min_def.cpp without outputs. | ||
| 21 | - All the input Tensors are numpy.ndarray. | ||
| 22 | - kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, | ||
| 23 | - input_formats, output_formats, input_ori_formats, output_ori_formats, | ||
| 24 | - input_dtypes, output_dtypes. | ||
| 25 | """ | 64 | """ |
| 26 | - import numpy as np | 65 | + if isinstance(dim, (tuple, list)): |
| 27 | - | 66 | + min_val = torch.amin(self, dim=dim, keepdim=bool(keepDim)) |
| 28 | - input_dtype = x.dtype | 67 | + max_val = torch.amax(self, dim=dim, keepdim=bool(keepDim)) |
| 29 | - if str(input_dtype) == "bfloat16": | ||
| 30 | - x = x.astype(np.float32) | ||
| 31 | - | ||
| 32 | - if axes is not None: | ||
| 33 | - axis = tuple(int(a) for a in np.asarray(axes).flatten()) | ||
| 34 | else: | 68 | else: |
| 35 | - axis = None | 69 | + result = torch.aminmax(self, dim=dim[0], keepdim=bool(keepDim)) |
| 70 | + min_val = result.min | ||
| 71 | + max_val = result.max | ||
| 72 | + return [min_val, max_val] | ||
| 36 | 73 | ||
| 37 | - # Try TensorFlow first (supports empty tensors), fallback to PyTorch, then NumPy | ||
| 38 | - try: | ||
| 39 | - import tensorflow as tf | ||
| 40 | 74 | ||
| 41 | - x_tensor = tf.constant(x) | 75 | +def aclnn_aminmax_all_golden(self, minOut=None, maxOut=None, **kwargs): |
| 42 | - res_tensor = tf.reduce_min(x_tensor, axis=axis, keepdims=keep_dims) | 76 | + """ |
| 43 | - res = res_tensor.numpy() | 77 | + Aclnn golden for aclnnAminmaxAll. |
| 44 | - except ImportError: | 78 | + """ |
| 45 | - try: | 79 | + result = torch.aminmax(self) |
| 46 | - import torch | 80 | + return [result.min, result.max] |
| 47 | 81 | ||
| 48 | - x_torch = torch.from_numpy(x) | 82 | + |
| 49 | - if axis is not None: | 83 | +def aclnn_aminmax_dim_golden( |
| 50 | - res_torch = torch.amin(x_torch, dim=axis, keepdim=keep_dims) | 84 | + self, dim=0, keepDim=0, minOut=None, maxOut=None, **kwargs |
| 51 | - else: | 85 | +): |
| 52 | - res_torch = torch.amin(x_torch, keepdim=keep_dims) | 86 | + """ |
| 53 | - res = res_torch.numpy() | 87 | + Aclnn golden for aclnnAminmaxDim. |
| 54 | - except ImportError: | 88 | + """ |
| 55 | - # Fallback to NumPy | 89 | + result = torch.aminmax(self, dim=dim, keepdim=bool(keepDim)) |
| 56 | - res = np.min(x, axis=axis, keepdims=keep_dims) | 90 | + return [result.min, result.max] |
| 57 | - return res.astype(input_dtype, copy=False) | ||
| @@ -0,0 +1,2 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,attributes,tensor_view_shapes,output_tensor_indexes,input_data_ranges,precision_tolerances,absolute_precision | ||
| 2 | +aclnnAminmaxAll_00,aclnnAminmaxAll,"('float32', 'float32', 'float32')",,"((2,35),(1,),(1,))","(1,2)","[[0, 0.001]]","((0.001, 0.001),)",0.000001 | ||
| @@ -0,0 +1,2 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,attributes,tensor_view_shapes,output_tensor_indexes,input_data_ranges,precision_tolerances,absolute_precision | ||
| 2 | +aclnnAminmaxDim_00,aclnnAminmaxDim,"('float32', 'float32', 'float32')","{'dim': 4, 'keepDim': True}","((1,7831,1,1,107),(1,7831,1,1,1),(1,7831,1,1,1))","(1,2)","[[0, 0.001]]","((0.001, 0.001),)",0.000001 | ||
| @@ -0,0 +1,2 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,attributes,tensor_view_shapes,output_tensor_indexes,input_data_ranges,precision_tolerances,absolute_precision | ||
| 2 | +aclnnAminmax_00,aclnnAminmax,"('float32', 'float32', 'float32')","{'dim': [4,], 'keepDim': True}","((1,7831,1,1,107),(1,7831,1,1,1),(1,7831,1,1,1))","(1,2)","[[0, 0.001]]","((0.001, 0.001),)",0.000001 | ||
| @@ -4,7 +4,7 @@ | |||
| 4 | # Copyright (c) 2026 Huawei Technologies Co., Ltd. | 4 | # Copyright (c) 2026 Huawei Technologies Co., Ltd. |
| 5 | # This program is free software, you can redistribute it and/or modify it under the terms and conditions of | 5 | # This program is free software, you can redistribute it and/or modify it under the terms and conditions of |
| 6 | # CANN Open Software License Agreement Version 2.0 (the "License"). | 6 | # CANN Open Software License Agreement Version 2.0 (the "License"). |
| 7 | -# Please refer to the License for details. You may not use your file except compliance with the License. | 7 | +# Please refer to the License for details. You may not use this file except in compliance with the License. |
| 8 | # THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, | 8 | # THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, |
| 9 | # INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. | 9 | # INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. |
| 10 | # See LICENSE in the root of the software repository for the full text of the License. | 10 | # See LICENSE in the root of the software repository for the full text of the License. |
| @@ -14,26 +14,28 @@ import numpy as np | |||
| 14 | 14 | ||
| 15 | 15 | ||
| 16 | __golden__ = { | 16 | __golden__ = { |
| 17 | - "kernel": { | 17 | + "aclnn": { |
| 18 | - "rsqrt": "rsqrt_golden" | 18 | + "aclnnInplaceRsqrt": "aclnn_inplace_rsqrt_golden", |
| 19 | - } | 19 | + "aclnnRsqrt": "aclnn_rsqrt_golden", |
| 20 | + }, | ||
| 21 | + "kernel": {"rsqrt": "rsqrt_golden"}, | ||
| 20 | } | 22 | } |
| 21 | 23 | ||
| 22 | 24 | ||
| 23 | def rsqrt_golden(x, **kwargs): | 25 | def rsqrt_golden(x, **kwargs): |
| 24 | - ''' | 26 | + """ |
| 25 | Kernel golden for rsqrt. | 27 | Kernel golden for rsqrt. |
| 26 | All the parameters follow @rsqrt_def.cpp without outputs. | 28 | All the parameters follow @rsqrt_def.cpp without outputs. |
| 27 | All the input Tensors are numpy.ndarray. | 29 | All the input Tensors are numpy.ndarray. |
| 28 | kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, | 30 | kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, |
| 29 | input_formats, output_formats, input_ori_formats, output_ori_formats, | 31 | input_formats, output_formats, input_ori_formats, output_ori_formats, |
| 30 | input_dtypes, output_dtypes. | 32 | input_dtypes, output_dtypes. |
| 31 | - ''' | 33 | + """ |
| 32 | import torch | 34 | import torch |
| 33 | - | 35 | + |
| 34 | ori_dtype = kwargs.get("input_dtypes", ["float32"])[0] | 36 | ori_dtype = kwargs.get("input_dtypes", ["float32"])[0] |
| 35 | x_dtype = x.dtype | 37 | x_dtype = x.dtype |
| 36 | - | 38 | + |
| 37 | if ori_dtype and "bfloat16" in str(ori_dtype).lower(): | 39 | if ori_dtype and "bfloat16" in str(ori_dtype).lower(): |
| 38 | x_tensor = torch.from_numpy(x.astype(np.float32)) | 40 | x_tensor = torch.from_numpy(x.astype(np.float32)) |
| 39 | output = torch.rsqrt(x_tensor) | 41 | output = torch.rsqrt(x_tensor) |
| @@ -45,4 +47,44 @@ def rsqrt_golden(x, **kwargs): | |||
| 45 | else: | 47 | else: |
| 46 | x_tensor = torch.from_numpy(x) | 48 | x_tensor = torch.from_numpy(x) |
| 47 | output = torch.rsqrt(x_tensor) | 49 | output = torch.rsqrt(x_tensor) |
| 48 | - return output.numpy() | 50 | + return output.numpy() |
| 51 | + | ||
| 52 | + | ||
| 53 | +def aclnn_rsqrt_golden(self, out=None, **kwargs): | ||
| 54 | + """ | ||
| 55 | + Aclnn golden for aclnnRsqrt. | ||
| 56 | + Parameters follow @aclnnRsqrtGetWorkspaceSize without workspaceSize & executor. | ||
| 57 | + All the input Tensors are torch.Tensor. | ||
| 58 | + """ | ||
| 59 | + import torch | ||
| 60 | + | ||
| 61 | + x = self | ||
| 62 | + x_dtype = x.dtype | ||
| 63 | + if x_dtype == torch.float16 or x_dtype == torch.bfloat16: | ||
| 64 | + y = torch.ops.aten.rsqrt(x.to(torch.float32)) | ||
| 65 | + else: | ||
| 66 | + y = torch.ops.aten.rsqrt(x) | ||
| 67 | + | ||
| 68 | + if x_dtype == torch.float16 or x_dtype == torch.bfloat16: | ||
| 69 | + y = y.to(x_dtype) | ||
| 70 | + return y | ||
| 71 | + | ||
| 72 | + | ||
| 73 | +def aclnn_inplace_rsqrt_golden(selfRef=None, **kwargs): | ||
| 74 | + """ | ||
| 75 | + Aclnn golden for aclnnInplaceRsqrt. | ||
| 76 | + Parameters follow @aclnnInplaceRsqrtGetWorkspaceSize without workspaceSize & executor. | ||
| 77 | + All the input Tensors are torch.Tensor. | ||
| 78 | + """ | ||
| 79 | + import torch | ||
| 80 | + | ||
| 81 | + x = selfRef | ||
| 82 | + x_dtype = x.dtype | ||
| 83 | + if x_dtype == torch.float16 or x_dtype == torch.bfloat16: | ||
| 84 | + y = torch.ops.aten.rsqrt(x.to(torch.float32)) | ||
| 85 | + else: | ||
| 86 | + y = torch.ops.aten.rsqrt(x) | ||
| 87 | + | ||
| 88 | + if x_dtype == torch.float16 or x_dtype == torch.bfloat16: | ||
| 89 | + y = y.to(x_dtype) | ||
| 90 | + return y | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,absolute_precision | ||
| 2 | +InplaceRsqrt_float32_ND_fuzz_1,aclnnInplaceRsqrt,('float32'),"('ND',)","((196, 2, 1, 76, 1, 1))","((0.001,1000),)","(-1,)","('float32',)",0.0001 | ||
| 3 | +InplaceRsqrt_float16_ND_fuzz_2,aclnnInplaceRsqrt,('float16'),"('ND',)","((192, 64, 1, 1, 1, 5, 1))","((0.001,1000),)","(-1,)","('float16',)",0.001 | ||
| 4 | +InplaceRsqrt_float32_ND_fuzz_3,aclnnInplaceRsqrt,('float32'),"('ND',)","((1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)","('float32',)",0.0001 | ||
| 5 | +InplaceRsqrt_float16_ND_fuzz_4,aclnnInplaceRsqrt,('float16'),"('ND',)","((139, 3, 1, 1, 1, 192))","((0.001,1000),)","(-1,)","('float16',)",0.001 | ||
| 6 | +InplaceRsqrt_float16_ND_fuzz_5,aclnnInplaceRsqrt,('float16'),"('ND',)","((1, 139, 1, 28, 1, 22, 1))","((0.001,1000),)","(-1,)","('float16',)",0.001 | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +,testcase_name,api_name,tensor_view_shapes,tensor_dtypes | ||
| 2 | +0,aclnnRsqrt_00000,aclnnRsqrt,"[[46, 40, 9, 3, 24, 36],[46, 40, 9, 3, 24, 36]]","('float32','float32')" | ||
| 3 | +1,aclnnRsqrt_00001,aclnnRsqrt,"[[34, 22, 39, 36],[34, 22, 39, 36]]","('bfloat16','bfloat16')" | ||
| 4 | +2,aclnnRsqrt_00002,aclnnRsqrt,"[[3],[3]]","('float32','float32')" | ||
| 5 | +3,aclnnRsqrt_00003,aclnnRsqrt,"[[17, 30, 49, 11],[17, 30, 49, 11]]","('float32','float32')" | ||
| 6 | +4,aclnnRsqrt_00004,aclnnRsqrt,"[[16, 25],[16, 25]]","('float16','float16')" | ||
| @@ -14,26 +14,27 @@ import numpy as np | |||
| 14 | 14 | ||
| 15 | 15 | ||
| 16 | __golden__ = { | 16 | __golden__ = { |
| 17 | - "kernel": { | 17 | + "aclnn": { |
| 18 | - "sign": "sign_golden" | 18 | + "aclnnSign": "aclnn_sign_golden", |
| 19 | - } | 19 | + }, |
| 20 | + "kernel": {"sign": "sign_golden"}, | ||
| 20 | } | 21 | } |
| 21 | 22 | ||
| 22 | 23 | ||
| 23 | def sign_golden(x, **kwargs): | 24 | def sign_golden(x, **kwargs): |
| 24 | - ''' | 25 | + """ |
| 25 | Kernel golden for sign. | 26 | Kernel golden for sign. |
| 26 | All the parameters follow @sign_def.cpp without outputs. | 27 | All the parameters follow @sign_def.cpp without outputs. |
| 27 | All the input Tensors are numpy.ndarray. | 28 | All the input Tensors are numpy.ndarray. |
| 28 | kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, | 29 | kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, |
| 29 | input_formats, output_formats, input_ori_formats, output_ori_formats, | 30 | input_formats, output_formats, input_ori_formats, output_ori_formats, |
| 30 | input_dtypes, output_dtypes. | 31 | input_dtypes, output_dtypes. |
| 31 | - ''' | 32 | + """ |
| 32 | import torch | 33 | import torch |
| 33 | - | 34 | + |
| 34 | ori_dtype = kwargs.get("input_dtypes", ["float32"])[0] | 35 | ori_dtype = kwargs.get("input_dtypes", ["float32"])[0] |
| 35 | x_dtype = x.dtype | 36 | x_dtype = x.dtype |
| 36 | - | 37 | + |
| 37 | if ori_dtype and "bfloat16" in str(ori_dtype).lower(): | 38 | if ori_dtype and "bfloat16" in str(ori_dtype).lower(): |
| 38 | x_tensor = torch.from_numpy(x.astype(np.float32)) | 39 | x_tensor = torch.from_numpy(x.astype(np.float32)) |
| 39 | output = torch.sign(x_tensor) | 40 | output = torch.sign(x_tensor) |
| @@ -45,4 +46,15 @@ def sign_golden(x, **kwargs): | |||
| 45 | else: | 46 | else: |
| 46 | x_tensor = torch.from_numpy(x) | 47 | x_tensor = torch.from_numpy(x) |
| 47 | output = torch.sign(x_tensor) | 48 | output = torch.sign(x_tensor) |
| 48 | - return output.numpy() | 49 | + return output.numpy() |
| 50 | + | ||
| 51 | + | ||
| 52 | +def aclnn_sign_golden(self, result=None, **kwargs): | ||
| 53 | + """ | ||
| 54 | + Aclnn golden for aclnnSign. | ||
| 55 | + Parameters follow @aclnnSignGetWorkspaceSize without workspaceSize & executor. | ||
| 56 | + All the input Tensors are torch.Tensor. | ||
| 57 | + """ | ||
| 58 | + import torch | ||
| 59 | + | ||
| 60 | + return [torch.sign(self)] | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +,testcase_name,api_name,tensor_view_shapes,tensor_dtypes | ||
| 2 | +0,aclnnSign_00000,aclnnSign,"[[23],[23]]","('float16','float16')" | ||
| 3 | +1,aclnnSign_00001,aclnnSign,"[[3, 11, 7],[3, 11, 7]]","('int32','int32')" | ||
| 4 | +2,aclnnSign_00002,aclnnSign,"[[26, 1, 42, 25, 45],[26, 1, 42, 25, 45]]","('int64','int64')" | ||
| 5 | +3,aclnnSign_00003,aclnnSign,"[[9, 17, 44],[9, 17, 44]]","('bfloat16','bfloat16')" | ||
| 6 | +4,aclnnSign_00004,aclnnSign,"[[22, 45, 19],[22, 45, 19]]","('float32','float32')" | ||
| @@ -11,29 +11,31 @@ | |||
| 11 | # ---------------------------------------------------------------------------- | 11 | # ---------------------------------------------------------------------------- |
| 12 | 12 | ||
| 13 | import numpy as np | 13 | import numpy as np |
| 14 | +import torch | ||
| 14 | 15 | ||
| 15 | 16 | ||
| 16 | __golden__ = { | 17 | __golden__ = { |
| 17 | - "kernel": { | 18 | + "aclnn": { |
| 18 | - "sin": "sin_golden" | 19 | + "aclnnSin": "aclnn_sin_golden", |
| 19 | - } | 20 | + }, |
| 21 | + "kernel": {"sin": "sin_golden"}, | ||
| 20 | } | 22 | } |
| 21 | 23 | ||
| 22 | 24 | ||
| 23 | def sin_golden(x, **kwargs): | 25 | def sin_golden(x, **kwargs): |
| 24 | - ''' | 26 | + """ |
| 25 | Kernel golden for sin. | 27 | Kernel golden for sin. |
| 26 | All the parameters follow @sin_def.cpp without outputs. | 28 | All the parameters follow @sin_def.cpp without outputs. |
| 27 | All the input Tensors are numpy.ndarray. | 29 | All the input Tensors are numpy.ndarray. |
| 28 | kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, | 30 | kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, |
| 29 | input_formats, output_formats, input_ori_formats, output_ori_formats, | 31 | input_formats, output_formats, input_ori_formats, output_ori_formats, |
| 30 | input_dtypes, output_dtypes. | 32 | input_dtypes, output_dtypes. |
| 31 | - ''' | 33 | + """ |
| 32 | import torch | 34 | import torch |
| 33 | - | 35 | + |
| 34 | ori_dtype = kwargs.get("input_dtypes", ["float32"])[0] | 36 | ori_dtype = kwargs.get("input_dtypes", ["float32"])[0] |
| 35 | x_dtype = x.dtype | 37 | x_dtype = x.dtype |
| 36 | - | 38 | + |
| 37 | if ori_dtype and "bfloat16" in str(ori_dtype).lower(): | 39 | if ori_dtype and "bfloat16" in str(ori_dtype).lower(): |
| 38 | x_tensor = torch.from_numpy(x.astype(np.float32)) | 40 | x_tensor = torch.from_numpy(x.astype(np.float32)) |
| 39 | output = torch.sin(x_tensor) | 41 | output = torch.sin(x_tensor) |
| @@ -43,4 +45,8 @@ def sin_golden(x, **kwargs): | |||
| 43 | output = torch.sin(x_tensor) | 45 | output = torch.sin(x_tensor) |
| 44 | return output.numpy().astype(x_dtype, copy=False) | 46 | return output.numpy().astype(x_dtype, copy=False) |
| 45 | else: | 47 | else: |
| 46 | - return np.sin(x) | 48 | + return np.sin(x) |
| 49 | + | ||
| 50 | + | ||
| 51 | +def aclnn_sin_golden(input, out=None, **kwargs): | ||
| 52 | + return [torch.sin(input)] | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,absolute_precision | ||
| 2 | +Sin_float32_ND_fuzz_1,aclnnSin,"('float32', 'float32')","('ND',)","((196, 2, 1, 76),(196, 2, 1, 76))","((-1000, -10),)","(-1,)","('float32',)",0.0001 | ||
| 3 | +Sin_float16_ND_fuzz_2,aclnnSin,"('float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('float16',)",0.001 | ||
| 4 | +Sin_float32_ND_fuzz_3,aclnnSin,"('float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)","('float32',)",0.0001 | ||
| 5 | +Sin_float16_ND_fuzz_4,aclnnSin,"('float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('float16',)",0.001 | ||
| 6 | +Sin_float16_ND_fuzz_5,aclnnSin,"('float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)","('float16',)",0.001 | ||
| @@ -0,0 +1,27 @@ | |||
| 1 | +#!/usr/bin/env python3 | ||
| 2 | +# -*- coding: UTF-8 -*- | ||
| 3 | +# ---------------------------------------------------------------------------- | ||
| 4 | +# Copyright (c) 2026 Huawei Technologies Co., Ltd. | ||
| 5 | +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of | ||
| 6 | +# CANN Open Software License Agreement Version 2.0 (the "License"). | ||
| 7 | +# Please refer to the License for details. You may not use this file except in compliance with the License. | ||
| 8 | +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, | ||
| 9 | +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. | ||
| 10 | +# See LICENSE in the root of the software repository for the full text of the License. | ||
| 11 | +# ---------------------------------------------------------------------------- | ||
| 12 | +import torch | ||
| 13 | + | ||
| 14 | +__golden__ = { | ||
| 15 | + "aclnn": { | ||
| 16 | + "aclnnSinh": "aclnn_sinh_golden", | ||
| 17 | + } | ||
| 18 | +} | ||
| 19 | + | ||
| 20 | + | ||
| 21 | +def aclnn_sinh_golden(self, out=None, **kwargs): | ||
| 22 | + """ | ||
| 23 | + Aclnn golden for aclnnSinh. | ||
| 24 | + Parameters follow @aclnnSinhGetWorkspaceSize without workspaceSize & executor. | ||
| 25 | + All the input Tensors are torch.Tensor. | ||
| 26 | + """ | ||
| 27 | + return [torch.sinh(self)] | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_view_shapes,tensor_dtypes,input_data_ranges,precision_tolerances | ||
| 2 | +aclnnSinh_fp32_1d_000001_0001,aclnnSinh,"((1,),(1,))","('float32','float32')","((-0.5,0.5),(-0.5,0.5))","(0.0001,0.0001)" | ||
| 3 | +aclnnSinh_fp32_1d_000002_0002,aclnnSinh,"((2,),(2,))","('float32','float32')","((-0.5,0.5),(-0.5,0.5))","(0.0001,0.0001)" | ||
| 4 | +aclnnSinh_fp32_1d_000003_0003,aclnnSinh,"((3,),(3,))","('float32','float32')","((-3.14,3.14),(-3.14,3.14))","(0.0001,0.0001)" | ||
| 5 | +aclnnSinh_fp32_1d_000004_0004,aclnnSinh,"((4,),(4,))","('float32','float32')","((-1,1),(-1,1))","(0.0001,0.0001)" | ||
| 6 | +aclnnSinh_fp32_1d_000005_0005,aclnnSinh,"((5,),(5,))","('float32','float32')","((-1,1),(-1,1))","(0.0001,0.0001)" | ||
| @@ -0,0 +1,35 @@ | |||
| 1 | +#!/usr/bin/env python3 | ||
| 2 | +# -*- coding: UTF-8 -*- | ||
| 3 | +# ---------------------------------------------------------------------------- | ||
| 4 | +# Copyright (c) 2026 Huawei Technologies Co., Ltd. | ||
| 5 | +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of | ||
| 6 | +# CANN Open Software License Agreement Version 2.0 (the "License"). | ||
| 7 | +# Please refer to the License for details. You may not use this file except in compliance with the License. | ||
| 8 | +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, | ||
| 9 | +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. | ||
| 10 | +# See LICENSE in the root of the software repository for the full text of the License. | ||
| 11 | +# ---------------------------------------------------------------------------- | ||
| 12 | +import torch | ||
| 13 | + | ||
| 14 | +__golden__ = { | ||
| 15 | + "aclnn": { | ||
| 16 | + "aclnnSort": "aclnn_sort_golden", | ||
| 17 | + } | ||
| 18 | +} | ||
| 19 | + | ||
| 20 | + | ||
| 21 | +def aclnn_sort_golden( | ||
| 22 | + self, stable=0, dim=0, descending=0, valuesOut=None, indicesOut=None, **kwargs | ||
| 23 | +): | ||
| 24 | + """ | ||
| 25 | + Aclnn golden for aclnnSort. | ||
| 26 | + Parameters follow @aclnnSortGetWorkspaceSize without workspaceSize & executor. | ||
| 27 | + All the input Tensors are torch.Tensor. | ||
| 28 | + """ | ||
| 29 | + input_x = self | ||
| 30 | + | ||
| 31 | + stable = stable | ||
| 32 | + dim = dim | ||
| 33 | + descending = descending | ||
| 34 | + y1, y2 = torch.sort(input=input_x, dim=dim, descending=descending, stable=True) | ||
| 35 | + return y1, y2 | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,attributes,tensor_view_shapes,output_tensor_indexes,input_data_ranges,is_enabled | ||
| 2 | +aclnn_sort_001,aclnnSort,"('float16','float16','int64')","('ND', 'ND', 'ND')","{ 'stable': True,'dim': -1, 'descending': True}","((1000, 14840), (1000, 14840), (1000,14840))","(1,2)","((-200,200),)",1 | ||
| 3 | +aclnn_sort_002,aclnnSort,"('bfloat16','bfloat16','int64')","('ND', 'ND', 'ND')","{ 'stable': True,'dim': -1, 'descending': False}","((1000, 14840), (1000, 14840), (1000,14840))","(1,2)","((-200,200),)",1 | ||
| 4 | +aclnn_sort_003,aclnnSort,"('float32','float32','int64')","('ND', 'ND', 'ND')","{ 'stable': True,'dim': -1, 'descending': False}","((3, 32768), (3, 32768), (3,32768))","(1,2)","((-200,200),)",1 | ||
| 5 | +aclnn_sort_004,aclnnSort,"('int32','int32','int64')","('ND', 'ND', 'ND')","{ 'stable': True,'dim': -1, 'descending': True}","((3, 32768), (3, 32768), (3,32768))","(1,2)","((-200,200),)",1 | ||
| 6 | +aclnn_sort_005,aclnnSort,"('int16','int16','int64')","('ND', 'ND', 'ND')","{ 'stable': True,'dim': -1, 'descending': False}","((3, 32768), (3, 32768), (3,32768))","(1,2)","((-200,200),)",1 | ||
| @@ -14,26 +14,27 @@ import numpy as np | |||
| 14 | 14 | ||
| 15 | 15 | ||
| 16 | __golden__ = { | 16 | __golden__ = { |
| 17 | - "kernel": { | 17 | + "aclnn": { |
| 18 | - "sqrt": "sqrt_golden" | 18 | + "aclnnSqrt": "aclnn_sqrt_golden", |
| 19 | - } | 19 | + }, |
| 20 | + "kernel": {"sqrt": "sqrt_golden"}, | ||
| 20 | } | 21 | } |
| 21 | 22 | ||
| 22 | 23 | ||
| 23 | def sqrt_golden(x, **kwargs): | 24 | def sqrt_golden(x, **kwargs): |
| 24 | - ''' | 25 | + """ |
| 25 | Kernel golden for sqrt. | 26 | Kernel golden for sqrt. |
| 26 | All the parameters follow @sqrt_def.cpp without outputs. | 27 | All the parameters follow @sqrt_def.cpp without outputs. |
| 27 | All the input Tensors are numpy.ndarray. | 28 | All the input Tensors are numpy.ndarray. |
| 28 | kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, | 29 | kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, |
| 29 | input_formats, output_formats, input_ori_formats, output_ori_formats, | 30 | input_formats, output_formats, input_ori_formats, output_ori_formats, |
| 30 | input_dtypes, output_dtypes. | 31 | input_dtypes, output_dtypes. |
| 31 | - ''' | 32 | + """ |
| 32 | import torch | 33 | import torch |
| 33 | - | 34 | + |
| 34 | ori_dtype = kwargs.get("input_dtypes", ["float32"])[0] | 35 | ori_dtype = kwargs.get("input_dtypes", ["float32"])[0] |
| 35 | x_dtype = x.dtype | 36 | x_dtype = x.dtype |
| 36 | - | 37 | + |
| 37 | if ori_dtype and "bfloat16" in str(ori_dtype).lower(): | 38 | if ori_dtype and "bfloat16" in str(ori_dtype).lower(): |
| 38 | x_tensor = torch.from_numpy(x.astype(np.float32)) | 39 | x_tensor = torch.from_numpy(x.astype(np.float32)) |
| 39 | result = torch.sqrt(x_tensor) | 40 | result = torch.sqrt(x_tensor) |
| @@ -45,4 +46,15 @@ def sqrt_golden(x, **kwargs): | |||
| 45 | else: | 46 | else: |
| 46 | x_tensor = torch.from_numpy(x) | 47 | x_tensor = torch.from_numpy(x) |
| 47 | result = torch.sqrt(x_tensor) | 48 | result = torch.sqrt(x_tensor) |
| 48 | - return result.numpy() | 49 | + return result.numpy() |
| 50 | + | ||
| 51 | + | ||
| 52 | +def aclnn_sqrt_golden(self, out=None, opExecutor=0, **kwargs): | ||
| 53 | + """ | ||
| 54 | + Aclnn golden for aclnnSqrt. | ||
| 55 | + Parameters follow @aclnnSqrtGetWorkspaceSize without workspaceSize & executor. | ||
| 56 | + All the input Tensors are torch.Tensor. | ||
| 57 | + """ | ||
| 58 | + import torch | ||
| 59 | + | ||
| 60 | + return [torch.sqrt(self)] | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,absolute_precision | ||
| 2 | +Sqrt_float32_ND_fuzz_1,aclnnSqrt,"('float32', 'float32')","('ND',)","((196, 2, 1, 76, 1, 1), (196, 2, 1, 76, 1, 1))","((-1000, -10),)","(-1,)",0.0001 | ||
| 3 | +Sqrt_float16_ND_fuzz_2,aclnnSqrt,"('float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)",0.001 | ||
| 4 | +Sqrt_float32_ND_fuzz_3,aclnnSqrt,"('float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)",0.0001 | ||
| 5 | +Sqrt_float16_ND_fuzz_4,aclnnSqrt,"('float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)",0.001 | ||
| 6 | +Sqrt_float16_ND_fuzz_5,aclnnSqrt,"('float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)",0.001 | ||
| @@ -15,19 +15,31 @@ import torch | |||
| 15 | 15 | ||
| 16 | 16 | ||
| 17 | __golden__ = { | 17 | __golden__ = { |
| 18 | - "kernel": { | 18 | + "aclnn": { |
| 19 | - "square": "square_golden" | 19 | + "aclnnSquare": "aclnn_square_golden", |
| 20 | - } | 20 | + }, |
| 21 | + "kernel": {"square": "square_golden"}, | ||
| 21 | } | 22 | } |
| 22 | 23 | ||
| 23 | 24 | ||
| 24 | def square_golden(x, **kwargs): | 25 | def square_golden(x, **kwargs): |
| 25 | ori_dtype = kwargs.get("input_dtypes", ["float32"])[0] | 26 | ori_dtype = kwargs.get("input_dtypes", ["float32"])[0] |
| 26 | x_dtype = x.dtype | 27 | x_dtype = x.dtype |
| 27 | - | 28 | + |
| 28 | if "bfloat16" in str(ori_dtype).lower() or "float16" in str(ori_dtype).lower(): | 29 | if "bfloat16" in str(ori_dtype).lower() or "float16" in str(ori_dtype).lower(): |
| 29 | x_tensor = torch.from_numpy(x.astype(np.float32)) | 30 | x_tensor = torch.from_numpy(x.astype(np.float32)) |
| 30 | output = torch.square(x_tensor) | 31 | output = torch.square(x_tensor) |
| 31 | return output.numpy().astype(x_dtype, copy=False) | 32 | return output.numpy().astype(x_dtype, copy=False) |
| 32 | else: | 33 | else: |
| 33 | - return np.square(x) | 34 | + return np.square(x) |
| 35 | + | ||
| 36 | + | ||
| 37 | +def aclnn_square_golden(self, out=None, **kwargs): | ||
| 38 | + """ | ||
| 39 | + Aclnn golden for aclnnSquare. | ||
| 40 | + Parameters follow @aclnnSquareGetWorkspaceSize without workspaceSize & executor. | ||
| 41 | + All the input Tensors are torch.Tensor. | ||
| 42 | + """ | ||
| 43 | + import torch | ||
| 44 | + | ||
| 45 | + return [torch.square(self)] | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,absolute_precision | ||
| 2 | +Square_float32_ND_fuzz_1,aclnnSquare,"('float32', 'float32')","('ND',)","((196, 2, 1, 76),(196, 2, 1, 76))","((-1000, -10),)","(-1,)","('float32',)",0.0001 | ||
| 3 | +Square_float16_ND_fuzz_2,aclnnSquare,"('float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('float16',)",0.001 | ||
| 4 | +Square_float32_ND_fuzz_3,aclnnSquare,"('float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)","('float32',)",0.0001 | ||
| 5 | +Square_float16_ND_fuzz_4,aclnnSquare,"('float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('float16',)",0.001 | ||
| 6 | +Square_float16_ND_fuzz_5,aclnnSquare,"('float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)","('float16',)",0.001 | ||
| @@ -0,0 +1,31 @@ | |||
| 1 | +#!/usr/bin/env python3 | ||
| 2 | +# -*- coding: UTF-8 -*- | ||
| 3 | +# ---------------------------------------------------------------------------- | ||
| 4 | +# Copyright (c) 2026 Huawei Technologies Co., Ltd. | ||
| 5 | +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of | ||
| 6 | +# CANN Open Software License Agreement Version 2.0 (the "License"). | ||
| 7 | +# Please refer to the License for details. You may not use this file except in compliance with the License. | ||
| 8 | +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, | ||
| 9 | +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. | ||
| 10 | +# See LICENSE in the root of the software repository for the full text of the License. | ||
| 11 | +# ---------------------------------------------------------------------------- | ||
| 12 | +import torch | ||
| 13 | + | ||
| 14 | +__golden__ = { | ||
| 15 | + "aclnn": { | ||
| 16 | + "aclnnTanh": "aclnn_tanh_golden", | ||
| 17 | + } | ||
| 18 | +} | ||
| 19 | + | ||
| 20 | + | ||
| 21 | +def aclnn_tanh_golden(self, out=None, **kwargs): | ||
| 22 | + """ | ||
| 23 | + Aclnn golden for aclnnTanh. | ||
| 24 | + Parameters follow @aclnnTanhGetWorkspaceSize without workspaceSize & executor. | ||
| 25 | + All the input Tensors are torch.Tensor. | ||
| 26 | + """ | ||
| 27 | + orig_dtype = self.dtype | ||
| 28 | + if orig_dtype in (torch.float16, torch.bfloat16): | ||
| 29 | + self = self.to(torch.float32) | ||
| 30 | + result = torch.tanh(self) | ||
| 31 | + return [result.to(orig_dtype)] | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,absolute_precision | ||
| 2 | +Tanh_float32_ND_fuzz_1,aclnnTanh,"('float32', 'float32')","('ND',)","((196, 2, 1, 76),(196, 2, 1, 76))","((-1000, -10),)","(-1,)","('float32',)",0.0001 | ||
| 3 | +Tanh_float16_ND_fuzz_2,aclnnTanh,"('float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('float16',)",0.001 | ||
| 4 | +Tanh_float32_ND_fuzz_3,aclnnTanh,"('float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)","('float32',)",0.0001 | ||
| 5 | +Tanh_float16_ND_fuzz_4,aclnnTanh,"('float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('float16',)",0.001 | ||
| 6 | +Tanh_float16_ND_fuzz_5,aclnnTanh,"('float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)","('float16',)",0.001 | ||
| @@ -11,35 +11,41 @@ | |||
| 11 | # ---------------------------------------------------------------------------- | 11 | # ---------------------------------------------------------------------------- |
| 12 | 12 | ||
| 13 | import numpy as np | 13 | import numpy as np |
| 14 | +import torch | ||
| 14 | 15 | ||
| 15 | 16 | ||
| 16 | __golden__ = { | 17 | __golden__ = { |
| 17 | - "kernel": { | 18 | + "aclnn": { |
| 18 | - "tensor_equal": "tensor_equal_golden" | 19 | + "aclnnEqual": "aclnn_equal_golden", |
| 19 | - } | 20 | + }, |
| 21 | + "kernel": {"tensor_equal": "tensor_equal_golden"}, | ||
| 20 | } | 22 | } |
| 21 | 23 | ||
| 22 | 24 | ||
| 23 | def tensor_equal_golden(input_x, input_y, **kwargs): | 25 | def tensor_equal_golden(input_x, input_y, **kwargs): |
| 24 | - ''' | 26 | + """ |
| 25 | Kernel golden for tensor_equal. | 27 | Kernel golden for tensor_equal. |
| 26 | All the parameters follow @tensor_equal_def.cpp without outputs. | 28 | All the parameters follow @tensor_equal_def.cpp without outputs. |
| 27 | All the input Tensors are numpy.ndarray. | 29 | All the input Tensors are numpy.ndarray. |
| 28 | kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, | 30 | kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, |
| 29 | input_formats, output_formats, input_ori_formats, output_ori_formats, | 31 | input_formats, output_formats, input_ori_formats, output_ori_formats, |
| 30 | input_dtypes, output_dtypes. | 32 | input_dtypes, output_dtypes. |
| 31 | - ''' | 33 | + """ |
| 32 | import torch | 34 | import torch |
| 33 | - | 35 | + |
| 34 | x_dtype = input_x.dtype | 36 | x_dtype = input_x.dtype |
| 35 | y_dtype = input_y.dtype | 37 | y_dtype = input_y.dtype |
| 36 | - | 38 | + |
| 37 | if str(x_dtype) == "bfloat16" and str(y_dtype) == "bfloat16": | 39 | if str(x_dtype) == "bfloat16" and str(y_dtype) == "bfloat16": |
| 38 | input_x = input_x.astype(np.float32) | 40 | input_x = input_x.astype(np.float32) |
| 39 | input_y = input_y.astype(np.float32) | 41 | input_y = input_y.astype(np.float32) |
| 40 | - | 42 | + |
| 41 | tensor_x = torch.tensor(input_x) | 43 | tensor_x = torch.tensor(input_x) |
| 42 | tensor_y = torch.tensor(input_y) | 44 | tensor_y = torch.tensor(input_y) |
| 43 | result = torch.equal(tensor_x, tensor_y) | 45 | result = torch.equal(tensor_x, tensor_y) |
| 44 | - | 46 | + |
| 45 | - return np.array(result) | 47 | + return np.array(result) |
| 48 | + | ||
| 49 | + | ||
| 50 | +def aclnn_equal_golden(self, other, out=None, **kwargs): | ||
| 51 | + return [torch.tensor(torch.equal(self, other))] | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_view_shapes,tensor_formats,tensor_dtypes,input_data_ranges,is_enabled | ||
| 2 | +aclnnEqual_test000001,aclnnEqual,"((229,3751),(229,3751),(1,),)","('ND',)","('float16', 'float16', 'bool' )","((2, 10), (-1, 0))",TRUE | ||
| 3 | +aclnnEqual_test000002,aclnnEqual,"((5743,746,6),(5743,746,6),(1,),)","('ND',)","('int8', 'int8', 'bool' )","((-2, -1), (-10, -2))",TRUE | ||
| 4 | +aclnnEqual_test000003,aclnnEqual,"((3,52,5,3591),(3,52,5,3591),(1,),)","('ND',)","('int16', 'int16', 'bool' )","((-1, 1), (-100, 100))",TRUE | ||
| 5 | +aclnnEqual_test000004,aclnnEqual,"((12,7,14,7,21),(12,7,14,7,21),(1,),)","('ND',)","('uint8', 'uint8', 'bool' )","((-10, 10), (0, 0))",TRUE | ||
| 6 | +aclnnEqual_test000005,aclnnEqual,"((6,4,15,1,7,18),(6,4,15,1,7,18),(1,),)","('ND',)","('uint16', 'uint16', 'bool' )","((0, 127), (127, 127))",TRUE | ||
| @@ -14,24 +14,32 @@ import numpy as np | |||
| 14 | 14 | ||
| 15 | 15 | ||
| 16 | __golden__ = { | 16 | __golden__ = { |
| 17 | - "kernel": { | 17 | + "aclnn": { |
| 18 | - "tile": "tile_golden", | 18 | + "aclnnRepeat": "aclnn_repeat_golden", |
| 19 | - "tile_d": "tile_golden" | 19 | + }, |
| 20 | - } | 20 | + "kernel": {"tile": "tile_golden", "tile_d": "tile_golden"}, |
| 21 | } | 21 | } |
| 22 | 22 | ||
| 23 | 23 | ||
| 24 | def tile_golden(x, multiples, **kwargs): | 24 | def tile_golden(x, multiples, **kwargs): |
| 25 | - ''' | 25 | + """ |
| 26 | Kernel golden for tile / tile_d. | 26 | Kernel golden for tile / tile_d. |
| 27 | All the parameters follow @tile_def.cpp without outputs. | 27 | All the parameters follow @tile_def.cpp without outputs. |
| 28 | All the input Tensors are numpy.ndarray. | 28 | All the input Tensors are numpy.ndarray. |
| 29 | kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, | 29 | kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, |
| 30 | input_formats, output_formats, input_ori_formats, output_ori_formats, | 30 | input_formats, output_formats, input_ori_formats, output_ori_formats, |
| 31 | input_dtypes, output_dtypes. | 31 | input_dtypes, output_dtypes. |
| 32 | - ''' | 32 | + """ |
| 33 | multiples_arr = np.array(multiples).astype(np.int64) | 33 | multiples_arr = np.array(multiples).astype(np.int64) |
| 34 | multiples_val = multiples_arr.tolist() | 34 | multiples_val = multiples_arr.tolist() |
| 35 | if isinstance(multiples_val, int): | 35 | if isinstance(multiples_val, int): |
| 36 | multiples_val = (multiples_val,) | 36 | multiples_val = (multiples_val,) |
| 37 | return np.tile(x, multiples_val) | 37 | return np.tile(x, multiples_val) |
| 38 | + | ||
| 39 | + | ||
| 40 | +def aclnn_repeat_golden(self, repeats=0, out=None, **kwargs): | ||
| 41 | + if hasattr(repeats, "tolist"): | ||
| 42 | + repeats = repeats.tolist() | ||
| 43 | + elif isinstance(repeats, int): | ||
| 44 | + repeats = [repeats] | ||
| 45 | + return [self.repeat(*repeats)] | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_view_shapes,tensor_dtypes,attributes,output_tensor_indexes,precision_tolerances | ||
| 2 | +tile_001,aclnnRepeat,"((2, 128, 1, 1),(2, 128, 28, 28))","('bfloat16','bfloat16')","{'repeats': [1, 1, 28, 28]}","(1,)", | ||
| 3 | +tile_002,aclnnRepeat,"((2, 128, 1, 1),(2, 128, 28, 28))","('bool','bool')","{'repeats': [1, 1, 28, 28]}","(1,)", | ||
| 4 | +tile_003,aclnnRepeat,"((1, 1, 32, 1, 1),(1, 43, 32, 1, 1))","('float16','float16')","{'repeats': [1, 43, 1, 1, 1]}","(1,)", | ||
| 5 | +tile_004,aclnnRepeat,"((2, 128, 28, 1, 1),(2, 128, 28, 28, 1))","('float32','float32')","{'repeats': [1, 1, 1, 28, 1]}","(1,)", | ||
| 6 | +tile_005,aclnnRepeat,"((2, 128, 1, 1),(2, 128, 28, 28))","('int32','int32')","{'repeats': [1, 1, 28, 28]}","(1,)", | ||
| @@ -14,26 +14,27 @@ import numpy as np | |||
| 14 | 14 | ||
| 15 | 15 | ||
| 16 | __golden__ = { | 16 | __golden__ = { |
| 17 | - "kernel": { | 17 | + "aclnn": { |
| 18 | - "trunc": "trunc_golden" | 18 | + "aclnnTrunc": "aclnn_trunc_golden", |
| 19 | - } | 19 | + }, |
| 20 | + "kernel": {"trunc": "trunc_golden"}, | ||
| 20 | } | 21 | } |
| 21 | 22 | ||
| 22 | 23 | ||
| 23 | def trunc_golden(input_x, **kwargs): | 24 | def trunc_golden(input_x, **kwargs): |
| 24 | - ''' | 25 | + """ |
| 25 | Kernel golden for trunc. | 26 | Kernel golden for trunc. |
| 26 | All the parameters follow @trunc_def.cpp without outputs. | 27 | All the parameters follow @trunc_def.cpp without outputs. |
| 27 | All the input Tensors are numpy.ndarray. | 28 | All the input Tensors are numpy.ndarray. |
| 28 | kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, | 29 | kwargs may contain: short_soc_version, input_ori_shapes, output_ori_shapes, |
| 29 | input_formats, output_formats, input_ori_formats, output_ori_formats, | 30 | input_formats, output_formats, input_ori_formats, output_ori_formats, |
| 30 | input_dtypes, output_dtypes. | 31 | input_dtypes, output_dtypes. |
| 31 | - ''' | 32 | + """ |
| 32 | import torch | 33 | import torch |
| 33 | - | 34 | + |
| 34 | ori_dtype = kwargs.get("input_dtypes", ["float32"])[0] | 35 | ori_dtype = kwargs.get("input_dtypes", ["float32"])[0] |
| 35 | x_dtype = input_x.dtype | 36 | x_dtype = input_x.dtype |
| 36 | - | 37 | + |
| 37 | if ori_dtype and "bfloat16" in str(ori_dtype).lower(): | 38 | if ori_dtype and "bfloat16" in str(ori_dtype).lower(): |
| 38 | x_tensor = torch.from_numpy(input_x.astype(np.float32)) | 39 | x_tensor = torch.from_numpy(input_x.astype(np.float32)) |
| 39 | output = torch.trunc(x_tensor) | 40 | output = torch.trunc(x_tensor) |
| @@ -45,4 +46,15 @@ def trunc_golden(input_x, **kwargs): | |||
| 45 | else: | 46 | else: |
| 46 | x_tensor = torch.from_numpy(input_x) | 47 | x_tensor = torch.from_numpy(input_x) |
| 47 | output = torch.trunc(x_tensor) | 48 | output = torch.trunc(x_tensor) |
| 48 | - return output.numpy() | 49 | + return output.numpy() |
| 50 | + | ||
| 51 | + | ||
| 52 | +def aclnn_trunc_golden(self, out=None, **kwargs): | ||
| 53 | + """ | ||
| 54 | + Aclnn golden for aclnnTrunc. | ||
| 55 | + Parameters follow @aclnnTruncGetWorkspaceSize without workspaceSize & executor. | ||
| 56 | + All the input Tensors are torch.Tensor. | ||
| 57 | + """ | ||
| 58 | + import torch | ||
| 59 | + | ||
| 60 | + return [torch.trunc(self)] | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,absolute_precision | ||
| 2 | +Trunc_float32_ND_fuzz_1,aclnnTrunc,"('float32', 'float32')","('ND',)","((196, 2, 1, 76),(196, 2, 1, 76))","((-1000, -10),)","(-1,)","('float32',)",0.0001 | ||
| 3 | +Trunc_float16_ND_fuzz_2,aclnnTrunc,"('float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('float16',)",0.001 | ||
| 4 | +Trunc_float32_ND_fuzz_3,aclnnTrunc,"('float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)","('float32',)",0.0001 | ||
| 5 | +Trunc_float16_ND_fuzz_4,aclnnTrunc,"('float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('float16',)",0.001 | ||
| 6 | +Trunc_float16_ND_fuzz_5,aclnnTrunc,"('float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)","('float16',)",0.001 | ||
| @@ -0,0 +1,239 @@ | |||
| 1 | +#!/usr/bin/env python3 | ||
| 2 | +# -*- coding: UTF-8 -*- | ||
| 3 | +# ---------------------------------------------------------------------------- | ||
| 4 | +# Copyright (c) 2026 Huawei Technologies Co., Ltd. | ||
| 5 | +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of | ||
| 6 | +# CANN Open Software License Agreement Version 2.0 (the "License"). | ||
| 7 | +# Please refer to the License for details. You may not use this file except in compliance with the License. | ||
| 8 | +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, | ||
| 9 | +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. | ||
| 10 | +# See LICENSE in the root of the software repository for the full text of the License. | ||
| 11 | +# ---------------------------------------------------------------------------- | ||
| 12 | +import numpy as np | ||
| 13 | +import torch | ||
| 14 | +from functools import reduce | ||
| 15 | + | ||
| 16 | +PHILOX_W32_0 = 0x9E3779B9 | ||
| 17 | +PHILOX_W32_1 = 0xBB67AE85 | ||
| 18 | +PHILOX_M4x32_0 = 0xD2511F53 | ||
| 19 | +PHILOX_M4x32_1 = 0xCD9E8D57 | ||
| 20 | +CURAND_2POW32_INV = 2.3283064e-10 | ||
| 21 | +CURAND_2POW32_INV_2PI = 2.3283064e-10 * 6.2831855 | ||
| 22 | +MAX_THREADS_PER_PROCESSOR = 2048 | ||
| 23 | +MAX_BLOCK_NUMS = 2147483647 | ||
| 24 | +PHILOX_BLOCK_THREAD = 512 | ||
| 25 | +STEP = 4 | ||
| 26 | + | ||
| 27 | + | ||
| 28 | +def mulhilo32(a, b): | ||
| 29 | + product = np.uint64(a) * b.astype(np.uint64) | ||
| 30 | + hi = (product >> 32) & 0xFFFFFFFF | ||
| 31 | + lo = product & 0xFFFFFFFF | ||
| 32 | + return hi.astype(np.uint32), lo.astype(np.uint32) | ||
| 33 | + | ||
| 34 | + | ||
| 35 | +def _philox4x32round(ctr, key): | ||
| 36 | + hi0, lo0 = mulhilo32(PHILOX_M4x32_0, ctr[0]) | ||
| 37 | + hi1, lo1 = mulhilo32(PHILOX_M4x32_1, ctr[2]) | ||
| 38 | + return np.array( | ||
| 39 | + [hi1 ^ ctr[1] ^ key[0], lo1, hi0 ^ ctr[3] ^ key[1], lo0], dtype=np.uint32 | ||
| 40 | + ) | ||
| 41 | + | ||
| 42 | + | ||
| 43 | +def rand_Philox4x32_10(c, k): | ||
| 44 | + k = k.copy() | ||
| 45 | + for i in range(9): | ||
| 46 | + c = _philox4x32round(c, k) | ||
| 47 | + k[0] += PHILOX_W32_0 | ||
| 48 | + k[1] += PHILOX_W32_1 | ||
| 49 | + k[0] &= 0xFFFFFFFF | ||
| 50 | + k[1] &= 0xFFFFFFFF | ||
| 51 | + return _philox4x32round(c, k) | ||
| 52 | + | ||
| 53 | + | ||
| 54 | +class RandStatePhilox4_32_10: | ||
| 55 | + def __init__(self): | ||
| 56 | + self.ctr = np.array([0, 0, 0, 0], dtype=np.uint32) | ||
| 57 | + self.key = np.array([0, 0], dtype=np.uint32) | ||
| 58 | + self.STATE = 0 | ||
| 59 | + self.output = np.array([0, 0, 0, 0], dtype=np.uint32) | ||
| 60 | + | ||
| 61 | + | ||
| 62 | +def Philox_State_Incr_hi(s, n): | ||
| 63 | + nlo = n & 0xFFFFFFFF | ||
| 64 | + nhi = (n >> 32) & 0xFFFFFFFF | ||
| 65 | + s.ctr[2] += nlo | ||
| 66 | + if s.ctr[2] < nlo: | ||
| 67 | + nhi += 1 | ||
| 68 | + s.ctr[3] += nhi | ||
| 69 | + s.ctr[3] &= 0xFFFFFFFF | ||
| 70 | + | ||
| 71 | + | ||
| 72 | +def Philox_State_Incr(s, n=None): | ||
| 73 | + if n is None: | ||
| 74 | + for i in range(STEP): | ||
| 75 | + s.ctr[i] += 1 | ||
| 76 | + if s.ctr[i] != 0: | ||
| 77 | + break | ||
| 78 | + else: | ||
| 79 | + nlo = n & 0xFFFFFFFF | ||
| 80 | + nhi = (n >> 32) & 0xFFFFFFFF | ||
| 81 | + s.ctr[0] += nlo | ||
| 82 | + if s.ctr[0] < nlo: | ||
| 83 | + nhi += 1 | ||
| 84 | + s.ctr[1] += nhi | ||
| 85 | + if s.ctr[1] < nhi: | ||
| 86 | + nhi += 1 | ||
| 87 | + s.ctr[2] += nhi | ||
| 88 | + if s.ctr[2] < nhi: | ||
| 89 | + nhi += 1 | ||
| 90 | + s.ctr[3] += nhi | ||
| 91 | + s.ctr &= 0xFFFFFFFF | ||
| 92 | + | ||
| 93 | + | ||
| 94 | +def skipahead_sequence(n, state): | ||
| 95 | + Philox_State_Incr_hi(state, n) | ||
| 96 | + state.output = rand_Philox4x32_10(state.ctr, state.key) | ||
| 97 | + | ||
| 98 | + | ||
| 99 | +def skipahead(n, state): | ||
| 100 | + state.STATE += n % 4 | ||
| 101 | + n //= 4 | ||
| 102 | + if state.STATE > 3: | ||
| 103 | + n += 1 | ||
| 104 | + state.STATE -= 4 | ||
| 105 | + Philox_State_Incr(state, n) | ||
| 106 | + state.output = rand_Philox4x32_10(state.ctr, state.key) | ||
| 107 | + | ||
| 108 | + | ||
| 109 | +def rand_init(seed, subsequence, offset, state): | ||
| 110 | + state.ctr = np.array([0, 0, 0, 0], dtype=np.uint32) | ||
| 111 | + state.key[0] = seed & 0xFFFFFFFF | ||
| 112 | + state.key[1] = (seed >> 32) & 0xFFFFFFFF | ||
| 113 | + state.STATE = 0 | ||
| 114 | + skipahead_sequence(subsequence, state) | ||
| 115 | + skipahead(offset, state) | ||
| 116 | + | ||
| 117 | + | ||
| 118 | +def curand(state): | ||
| 119 | + r = state.output[state.STATE] | ||
| 120 | + state.STATE += 1 | ||
| 121 | + if state.STATE == 4: | ||
| 122 | + Philox_State_Incr(state) | ||
| 123 | + state.output = rand_Philox4x32_10(state.ctr, state.key) | ||
| 124 | + state.STATE = 0 | ||
| 125 | + return r | ||
| 126 | + | ||
| 127 | + | ||
| 128 | +def rand4(state): | ||
| 129 | + tmp = state.output.copy() | ||
| 130 | + Philox_State_Incr(state) | ||
| 131 | + state.output = rand_Philox4x32_10(state.ctr, state.key) | ||
| 132 | + if state.STATE == 0: | ||
| 133 | + return tmp | ||
| 134 | + r = np.zeros(STEP, dtype=np.uint32) | ||
| 135 | + if state.STATE == 1: | ||
| 136 | + r[0], r[1], r[2], r[3] = tmp[1], tmp[2], tmp[3], state.output[0] | ||
| 137 | + elif state.STATE == 2: | ||
| 138 | + r[0], r[1], r[2], r[3] = tmp[2], tmp[3], state.output[0], state.output[1] | ||
| 139 | + elif state.STATE == 3: | ||
| 140 | + r[0], r[1], r[2], r[3] = ( | ||
| 141 | + tmp[3], | ||
| 142 | + state.output[0], | ||
| 143 | + state.output[1], | ||
| 144 | + state.output[2], | ||
| 145 | + ) | ||
| 146 | + return r | ||
| 147 | + | ||
| 148 | + | ||
| 149 | +def rand_uniform4(state): | ||
| 150 | + x = rand4(state) | ||
| 151 | + y = np.array( | ||
| 152 | + [ | ||
| 153 | + x[0] * CURAND_2POW32_INV + (CURAND_2POW32_INV / 2), | ||
| 154 | + x[1] * CURAND_2POW32_INV + (CURAND_2POW32_INV / 2), | ||
| 155 | + x[2] * CURAND_2POW32_INV + (CURAND_2POW32_INV / 2), | ||
| 156 | + x[3] * CURAND_2POW32_INV + (CURAND_2POW32_INV / 2), | ||
| 157 | + ], | ||
| 158 | + dtype=np.float32, | ||
| 159 | + ) | ||
| 160 | + return y | ||
| 161 | + | ||
| 162 | + | ||
| 163 | +def bernhoulli_h20(seed, offset, prob, total_ele): | ||
| 164 | + result = np.zeros(total_ele, dtype=np.uint32) | ||
| 165 | + minBlockNums = ( | ||
| 166 | + total_ele + MAX_THREADS_PER_PROCESSOR - 1 | ||
| 167 | + ) // MAX_THREADS_PER_PROCESSOR | ||
| 168 | + minBlockNums = min(minBlockNums, MAX_BLOCK_NUMS) | ||
| 169 | + total_thread = minBlockNums * PHILOX_BLOCK_THREAD | ||
| 170 | + repeat_time = ((total_ele + STEP - 1) // STEP + total_thread - 1) // total_thread | ||
| 171 | + for i in range(total_thread): | ||
| 172 | + s = RandStatePhilox4_32_10() | ||
| 173 | + rand_init(seed, i, offset, s) | ||
| 174 | + for j in range(repeat_time): | ||
| 175 | + x = rand_uniform4(s) | ||
| 176 | + realIndex = i * STEP + total_thread * STEP * j | ||
| 177 | + if realIndex >= total_ele: | ||
| 178 | + break | ||
| 179 | + result[realIndex] = 1 if x[0] <= prob[realIndex] else 0 | ||
| 180 | + if realIndex + 1 >= total_ele: | ||
| 181 | + break | ||
| 182 | + result[realIndex + 1] = 1 if x[1] <= prob[realIndex + 1] else 0 | ||
| 183 | + if realIndex + 2 >= total_ele: | ||
| 184 | + break | ||
| 185 | + result[realIndex + 2] = 1 if x[2] <= prob[realIndex + 2] else 0 | ||
| 186 | + if realIndex + 3 >= total_ele: | ||
| 187 | + break | ||
| 188 | + result[realIndex + 3] = 1 if x[3] <= prob[realIndex + 3] else 0 | ||
| 189 | + return result | ||
| 190 | + | ||
| 191 | + | ||
| 192 | +class _NumpyBFloat16: | ||
| 193 | + def __new__(cls): | ||
| 194 | + return np.dtype("float32") | ||
| 195 | + | ||
| 196 | + | ||
| 197 | +def numpy_bfloat16(): | ||
| 198 | + return np.dtype("float32") | ||
| 199 | + | ||
| 200 | + | ||
| 201 | +__golden__ = { | ||
| 202 | + "aclnn": { | ||
| 203 | + "aclnnBernoulli": "aclnn_bernoulli_golden", | ||
| 204 | + } | ||
| 205 | +} | ||
| 206 | + | ||
| 207 | + | ||
| 208 | +def aclnn_bernoulli_golden(self, prob, seed=0, offset=0, out=None, **kwargs): | ||
| 209 | + """ | ||
| 210 | + Aclnn golden for aclnnBernoulli. | ||
| 211 | + Parameters follow @aclnnBernoulliGetWorkspaceSize without workspaceSize & executor. | ||
| 212 | + All the input Tensors are torch.Tensor. | ||
| 213 | + """ | ||
| 214 | + if hasattr(prob, "item"): | ||
| 215 | + prob_val = prob.item() | ||
| 216 | + else: | ||
| 217 | + prob_val = prob | ||
| 218 | + if hasattr(seed, "item"): | ||
| 219 | + seed = seed.item() | ||
| 220 | + if hasattr(offset, "item"): | ||
| 221 | + offset = offset.item() | ||
| 222 | + | ||
| 223 | + seed = np.int64(seed).astype(np.uint64) | ||
| 224 | + offset = np.int64(offset).astype(np.uint64) | ||
| 225 | + | ||
| 226 | + x_shape = list(self.shape) | ||
| 227 | + tol = reduce(lambda x, y: x * y, x_shape) if x_shape else 1 | ||
| 228 | + if tol == 0: | ||
| 229 | + return [torch.empty(x_shape), torch.empty(x_shape)] | ||
| 230 | + | ||
| 231 | + prob_arr = np.array([prob_val]) | ||
| 232 | + out_total_ele = int(reduce(lambda x, y: x * y, x_shape)) if x_shape else 1 | ||
| 233 | + if prob_arr.size == 1: | ||
| 234 | + prob_arr = np.broadcast_to(prob_arr, out_total_ele) | ||
| 235 | + | ||
| 236 | + result = bernhoulli_h20( | ||
| 237 | + seed=seed, offset=offset, prob=prob_arr, total_ele=out_total_ele | ||
| 238 | + ) | ||
| 239 | + return [torch.from_numpy(result).to(out.dtype)] | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_view_shapes,tensor_formats,tensor_dtypes,scalar_dtypes,attributes,output_tensor_indexes,precision_tolerances,absolute_precision,is_enabled | ||
| 2 | +aclnn_bernoulli_test_0000,aclnnBernoulli,"((6,),(6,))","(('ND', 'ND'))","('uint8', 'uint8')","('float32',)","{'prob': 0.5, 'seed': 67, 'offset': 256}","(-1,)","((0.0001, 0.0001),)",0,TRUE | ||
| 3 | +aclnn_bernoulli_test_0001,aclnnBernoulli,"((1, 1, 1, 1, 1, 1),(1, 1, 1, 1, 1, 1))","(('ND', 'ND'))","('float16', 'float16')","('float32',)","{'prob': 0.12345, 'seed': 1, 'offset': 400}","(-1,)","((0.001, 0.001),)",0,TRUE | ||
| 4 | +aclnn_bernoulli_test_0002,aclnnBernoulli,"((100, 101, 102),(100, 101, 102))","(('ND', 'ND'))","('float32', 'float32')","('float16',)","{'prob': 0.12345, 'seed': 1, 'offset': 128}","(-1,)","((0.0001, 0.0001),)",0,TRUE | ||
| 5 | +aclnn_bernoulli_test_0003,aclnnBernoulli,"((1024, 16, 8),(1024, 16, 8))","(('ND', 'ND'))","('int8', 'int8')","('float16',)","{'prob': 0.1111, 'seed': 100, 'offset': 64}","(-1,)","((0.001, 0.001),)",0,TRUE | ||
| 6 | +aclnn_bernoulli_test_0004,aclnnBernoulli,"((0, 16, 8, 1024),(0, 16, 8, 1024))","(('ND', 'ND'))","('int8', 'int8')","('float32',)","{'prob': 0.1, 'seed': 100, 'offset': 512}","(-1,)","((0.0001, 0.0001),)",0,TRUE | ||