已合并
feat: Add aclnn golden functions and test cases #9487
yanzhi2024创建于 8月29日
feat: Add aclnn golden functions and test cases #9487
已合并
共 48 个文件变更+803-118
| @@ -12,18 +12,18 @@ | |||
| 12 | import numpy as np | 12 | import numpy as np |
| 13 | 13 | ||
| 14 | __golden__ = { | 14 | __golden__ = { |
| 15 | - "kernel": { | 15 | + "kernel": {"elu": "elu_golden"}, |
| 16 | - "elu": "elu_golden" | ||
| 17 | - }, | ||
| 18 | "aclnn": { | 16 | "aclnn": { |
| 19 | "aclnnElu": "aclnn_elu_golden", | 17 | "aclnnElu": "aclnn_elu_golden", |
| 20 | - "aclnnInplaceElu": "aclnn_inplace_elu_golden" | 18 | + "aclnnInplaceElu": "aclnn_inplace_elu_golden", |
| 21 | - } | 19 | + }, |
| 22 | } | 20 | } |
| 23 | 21 | ||
| 24 | 22 | ||
| 25 | -def elu_golden(x, *, alpha: float = 1.0, scale: float = 1.0, input_scale: float = 1.0, **kwargs): | 23 | +def elu_golden( |
| 26 | - ''' | 24 | + x, *, alpha: float = 1.0, scale: float = 1.0, input_scale: float = 1.0, **kwargs |
| 25 | +): | ||
| 26 | + """ | ||
| 27 | Golden function for elu. | 27 | Golden function for elu. |
| 28 | All the parameters (names and order) follow @elu_def.cpp without outputs. | 28 | All the parameters (names and order) follow @elu_def.cpp without outputs. |
| 29 | All the input Tensors are numpy.ndarray. | 29 | All the input Tensors are numpy.ndarray. |
| @@ -34,33 +34,38 @@ def elu_golden(x, *, alpha: float = 1.0, scale: float = 1.0, input_scale: float | |||
| 34 | 34 | ||
| 35 | Returns: | 35 | Returns: |
| 36 | Output tensor | 36 | Output tensor |
| 37 | - ''' | 37 | + """ |
| 38 | import torch | 38 | import torch |
| 39 | 39 | ||
| 40 | x_dtype = x.dtype | 40 | x_dtype = x.dtype |
| 41 | if x_dtype.name in ("bfloat16", "float16"): | 41 | if x_dtype.name in ("bfloat16", "float16"): |
| 42 | x = x.astype(np.float32) | 42 | x = x.astype(np.float32) |
| 43 | - | 43 | + |
| 44 | x_torch = torch.from_numpy(x) | 44 | x_torch = torch.from_numpy(x) |
| 45 | - output_y = torch.ops.aten.elu(x_torch, alpha=alpha, scale=scale, input_scale=input_scale) | 45 | + output_y = torch.ops.aten.elu( |
| 46 | + x_torch, alpha=alpha, scale=scale, input_scale=input_scale | ||
| 47 | + ) | ||
| 46 | result = output_y.numpy() | 48 | result = output_y.numpy() |
| 47 | - | 49 | + |
| 48 | if x_dtype.name in ("bfloat16", "float16"): | 50 | if x_dtype.name in ("bfloat16", "float16"): |
| 49 | result = result.astype(x_dtype, copy=False) | 51 | result = result.astype(x_dtype, copy=False) |
| 50 | - | 52 | + |
| 51 | return result | 53 | return result |
| 52 | 54 | ||
| 53 | 55 | ||
| 54 | def _aclnn_elu_impl(input_tensor, alpha, scale, inputScale): | 56 | def _aclnn_elu_impl(input_tensor, alpha, scale, inputScale): |
| 55 | import torch | 57 | import torch |
| 58 | + | ||
| 56 | alpha_val = alpha.item() | 59 | alpha_val = alpha.item() |
| 57 | scale_val = scale.item() | 60 | scale_val = scale.item() |
| 58 | input_scale_val = inputScale.item() | 61 | input_scale_val = inputScale.item() |
| 59 | - return torch.ops.aten.elu(input_tensor, alpha=alpha_val, scale=scale_val, input_scale=input_scale_val) | 62 | + return torch.ops.aten.elu( |
| 63 | + input_tensor, alpha=alpha_val, scale=scale_val, input_scale=input_scale_val | ||
| 64 | + ) | ||
| 60 | 65 | ||
| 61 | 66 | ||
| 62 | -def aclnn_elu_golden(selfT, alpha, scale, inputScale, out, **kwargs): | 67 | +def aclnn_elu_golden(self, alpha, scale, inputScale, out=None, **kwargs): |
| 63 | - ''' | 68 | + """ |
| 64 | Aclnn golden for aclnnElu. | 69 | Aclnn golden for aclnnElu. |
| 65 | All the parameters (name & order) follow \ | 70 | All the parameters (name & order) follow \ |
| 66 | function `aclnnEluGetWorkspaceSize` in @aclnn_elu.h \ | 71 | function `aclnnEluGetWorkspaceSize` in @aclnn_elu.h \ |
| @@ -74,12 +79,12 @@ def aclnn_elu_golden(selfT, alpha, scale, inputScale, out, **kwargs): | |||
| 74 | 79 | ||
| 75 | Returns: | 80 | Returns: |
| 76 | Output tensors. | 81 | Output tensors. |
| 77 | - ''' | 82 | + """ |
| 78 | - return _aclnn_elu_impl(selfT, alpha, scale, inputScale) | 83 | + return _aclnn_elu_impl(self, alpha, scale, inputScale) |
| 79 | 84 | ||
| 80 | 85 | ||
| 81 | def aclnn_inplace_elu_golden(selfRef, alpha, scale, inputScale, **kwargs): | 86 | def aclnn_inplace_elu_golden(selfRef, alpha, scale, inputScale, **kwargs): |
| 82 | - ''' | 87 | + """ |
| 83 | Aclnn golden for aclnnInplaceElu. | 88 | Aclnn golden for aclnnInplaceElu. |
| 84 | All the parameters (name & order) follow \ | 89 | All the parameters (name & order) follow \ |
| 85 | function `aclnnInplaceEluGetWorkspaceSize` in @aclnn_elu.h \ | 90 | function `aclnnInplaceEluGetWorkspaceSize` in @aclnn_elu.h \ |
| @@ -93,5 +98,5 @@ def aclnn_inplace_elu_golden(selfRef, alpha, scale, inputScale, **kwargs): | |||
| 93 | 98 | ||
| 94 | Returns: | 99 | Returns: |
| 95 | Output tensors. | 100 | Output tensors. |
| 96 | - ''' | 101 | + """ |
| 97 | return _aclnn_elu_impl(selfRef, alpha, scale, inputScale) | 102 | return _aclnn_elu_impl(selfRef, alpha, scale, inputScale) |
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,scalar_dtypes,scalar_data_ranges,absolute_precision | ||
| 2 | +Elu_float32_ND_fuzz_1,aclnnElu,"('float32', 'float32')","('ND',)","((196, 2, 1, 76, 1, 1), (196, 2, 1, 76, 1, 1))","((-1000, -10),)","(-1,)","('float32', 'float32', 'float32')","(0, 1)",0.0001 | ||
| 3 | +Elu_float16_ND_fuzz_2,aclnnElu,"('float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('float16', 'float16', 'float16')","(0, 1)",0.001 | ||
| 4 | +Elu_float32_ND_fuzz_3,aclnnElu,"('float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)","('float32', 'float32', 'float32')","(0, 1)",0.0001 | ||
| 5 | +Elu_float16_ND_fuzz_4,aclnnElu,"('float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('float16', 'float16', 'float16')","(0, 1)",0.001 | ||
| 6 | +Elu_float16_ND_fuzz_5,aclnnElu,"('float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)","('float16', 'float16', 'float16')","(0, 1)",0.001 | ||
| @@ -11,8 +11,14 @@ | |||
| 11 | # ---------------------------------------------------------------------------- | 11 | # ---------------------------------------------------------------------------- |
| 12 | 12 | ||
| 13 | import numpy as np | 13 | import numpy as np |
| 14 | +import torch | ||
| 14 | 15 | ||
| 15 | -__golden__ = {"kernel": {"elu_grad_v2": "elu_grad_v2_golden"}} | 16 | +__golden__ = { |
| 17 | + "aclnn": { | ||
| 18 | + "aclnnEluBackward": "aclnn_elu_backward_golden", | ||
| 19 | + }, | ||
| 20 | + "kernel": {"elu_grad_v2": "elu_grad_v2_golden"}, | ||
| 21 | +} | ||
| 16 | 22 | ||
| 17 | 23 | ||
| 18 | def elu_grad_v2_golden( | 24 | def elu_grad_v2_golden( |
| @@ -54,3 +60,17 @@ def elu_grad_v2_golden( | |||
| 54 | if grads_dtype.name in ("bfloat16", "float16"): | 60 | if grads_dtype.name in ("bfloat16", "float16"): |
| 55 | result = result.astype(grads_dtype, copy=False) | 61 | result = result.astype(grads_dtype, copy=False) |
| 56 | return result | 62 | return result |
| 63 | + | ||
| 64 | + | ||
| 65 | +def aclnn_elu_backward_golden( | ||
| 66 | + gradOutput, alpha, scale, inputScale, isResult, selfOrResult, gradInput, **kwargs | ||
| 67 | +): | ||
| 68 | + alpha_val = alpha.item() if hasattr(alpha, "item") else alpha | ||
| 69 | + scale_val = scale.item() if hasattr(scale, "item") else scale | ||
| 70 | + input_scale_val = inputScale.item() if hasattr(inputScale, "item") else inputScale | ||
| 71 | + is_result = bool(isResult.item()) if hasattr(isResult, "item") else bool(isResult) | ||
| 72 | + return [ | ||
| 73 | + torch.ops.aten.elu_backward( | ||
| 74 | + gradOutput, alpha_val, scale_val, input_scale_val, is_result, selfOrResult | ||
| 75 | + ) | ||
| 76 | + ] | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,absolute_precision,attributes,scalar_dtypes,scalar_data_ranges | ||
| 2 | +EluGradV2_float32_ND_fuzz_1,aclnnEluBackward,"('float32', 'float32','float32')","('ND',)","((196, 2, 1, 76),(196, 2, 1, 76),(196, 2, 1, 76))","((-1000, -10),)","(-1,)","('float32',)",0.0001,{'isResult':True},"('float32', 'float32','float32')","((1,2),(3,4),(0,2))" | ||
| 3 | +EluGradV2_float16_ND_fuzz_2,aclnnEluBackward,"('float16', 'float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('float16',)",0.001,{'isResult':True},"('float32', 'float32','float32')","((0,2),(1,2),(0,2))" | ||
| 4 | +EluGradV2_float32_ND_fuzz_3,aclnnEluBackward,"('float32', 'float32','float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)","('float32',)",0.0001,{'isResult':False},"('float32', 'float32','float32')","((1,2),(3,4),(0,2))" | ||
| 5 | +EluGradV2_float16_ND_fuzz_4,aclnnEluBackward,"('float16', 'float16','float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192),(139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('float16',)",0.001,{'isResult':False},"('float32', 'float32','float32')","((0,2),(1,2),(0,2))" | ||
| 6 | +EluGradV2_float16_ND_fuzz_5,aclnnEluBackward,"('float16', 'float16','float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1),(1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)","('float16',)",0.001,{'isResult':True},"('float32', 'float32','float32')","((1,2),(3,4),(0,2))" | ||
| @@ -1,5 +1,5 @@ | |||
| 1 | #!/usr/bin/env python3 | 1 | #!/usr/bin/env python3 |
| 2 | -# -*- coding: UTF-8 -*- | 2 | +# -*- coding: utf-8 -*- |
| 3 | # ---------------------------------------------------------------------------- | 3 | # ---------------------------------------------------------------------------- |
| 4 | # Copyright (c) 2026 Huawei Technologies Co., Ltd. | 4 | # Copyright (c) 2026 Huawei Technologies Co., Ltd. |
| 5 | # This program is free software, you can redistribute it and/or modify it under the terms and conditions of | 5 | # This program is free software, you can redistribute it and/or modify it under the terms and conditions of |
| @@ -12,7 +12,10 @@ | |||
| 12 | 12 | ||
| 13 | import numpy as np | 13 | import numpy as np |
| 14 | 14 | ||
| 15 | -__golden__ = {"kernel": {"ge_glu_v2": "ge_glu_v2_golden"}} | 15 | +__golden__ = { |
| 16 | + "kernel": {"ge_glu_v2": "ge_glu_v2_golden"}, | ||
| 17 | + "aclnn": {"aclnnGeGlu": "aclnn_ge_glu_golden"}, | ||
| 18 | +} | ||
| 16 | 19 | ||
| 17 | 20 | ||
| 18 | def do_gelu(x, approximate): | 21 | def do_gelu(x, approximate): |
| @@ -37,8 +40,16 @@ def process(input_x, split_dim, activateLeft, approximate): | |||
| 37 | gate, x = tensor_x.chunk(2, dim=split_dim) | 40 | gate, x = tensor_x.chunk(2, dim=split_dim) |
| 38 | else: | 41 | else: |
| 39 | x, gate = tensor_x.chunk(2, dim=split_dim) | 42 | x, gate = tensor_x.chunk(2, dim=split_dim) |
| 43 | + | ||
| 44 | + if gate.dtype == torch.half: | ||
| 45 | + gate = gate.to(torch.float32) | ||
| 46 | + | ||
| 40 | y_gelu = do_gelu(gate, approximate) | 47 | y_gelu = do_gelu(gate, approximate) |
| 41 | - y = x * y_gelu | 48 | + |
| 49 | + if x.dtype == torch.half: | ||
| 50 | + y = x * y_gelu.to(torch.half) | ||
| 51 | + else: | ||
| 52 | + y = x * y_gelu | ||
| 42 | return y, y_gelu | 53 | return y, y_gelu |
| 43 | 54 | ||
| 44 | 55 | ||
| @@ -57,7 +68,35 @@ def ge_glu_v2_golden(x, *, dim=-1, approximate=1, activate_left=False, **kwargs) | |||
| 57 | """ | 68 | """ |
| 58 | dtype = x.dtype | 69 | dtype = x.dtype |
| 59 | if dtype != np.float64: | 70 | if dtype != np.float64: |
| 60 | - x = x.astype(np.float32) | 71 | + if dtype != np.float16 or approximate != 1: |
| 72 | + x = x.astype(np.float32) | ||
| 61 | 73 | ||
| 62 | y, y_gelu = process(x, dim, activate_left, approximate) | 74 | y, y_gelu = process(x, dim, activate_left, approximate) |
| 63 | return y.numpy().astype(dtype), y_gelu.numpy().astype(dtype) | 75 | return y.numpy().astype(dtype), y_gelu.numpy().astype(dtype) |
| 76 | + | ||
| 77 | + | ||
| 78 | +def aclnn_ge_glu_golden(self, dim, approximate, out=None, outGelu=None, **kwargs): | ||
| 79 | + """ | ||
| 80 | + Aclnn golden for aclnnGeGlu. | ||
| 81 | + Parameters follow @aclnnGeGluGetWorkspaceSize without workspaceSize & executor. | ||
| 82 | + All the input Tensors are torch.Tensor. | ||
| 83 | + """ | ||
| 84 | + import torch | ||
| 85 | + | ||
| 86 | + if hasattr(dim, "item"): | ||
| 87 | + dim = dim.item() | ||
| 88 | + if hasattr(approximate, "item"): | ||
| 89 | + approximate = approximate.item() | ||
| 90 | + | ||
| 91 | + orig_dtype = self.dtype | ||
| 92 | + input_x = self | ||
| 93 | + if orig_dtype != torch.float64: | ||
| 94 | + if orig_dtype != torch.float16 or approximate != 1: | ||
| 95 | + input_x = input_x.to(torch.float32) | ||
| 96 | + | ||
| 97 | + x_np = input_x.numpy() | ||
| 98 | + y, y_gelu = process(x_np, dim, False, approximate) | ||
| 99 | + | ||
| 100 | + y = y.to(orig_dtype) | ||
| 101 | + y_gelu = y_gelu.to(orig_dtype) | ||
| 102 | + return [y, y_gelu] | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,attributes,absolute_precision | ||
| 2 | +Geglu_float32_ND_fuzz_1,aclnnGeGlu,"('float32', 'float32', 'float32')","('ND',)","((1, 1, 1, 1, 1, 1, 2, 2), (1, 1, 1, 1, 1, 1, 1, 2), (1, 1, 1, 1, 1, 1, 1, 2))","((-1000, -10),)","(-2,-1)","{'dim': 6, 'approximate': 1}",0.0001 | ||
| 3 | +Geglu_float16_ND_fuzz_2,aclnnGeGlu,"('float16', 'float16', 'float16')","('ND',)","((2, 1, 271, 1, 1, 1, 2), (1, 1, 271, 1, 1, 1, 2), (1, 1, 271, 1, 1, 1, 2))","((0, 0),)","(-2,-1)","{'dim': 0, 'approximate': 1}",0.001 | ||
| 4 | +Geglu_float32_ND_fuzz_3,aclnnGeGlu,"('float32', 'float32', 'float32')","('ND',)","((1, 4), (1,2), (1,2))","((-3.4e+38, 3.4e+38),)","(-2,-1)","{'dim': -1, 'approximate': 1}",0.0001 | ||
| 5 | +Geglu_float16_ND_fuzz_4,aclnnGeGlu,"('float16', 'float16', 'float16')","('ND',)","((1, 3461, 172), (1, 3461, 86), (1, 3461, 86))","((-10, -2),)","(-2,-1)","{'dim': -1, 'approximate': 1}",0.001 | ||
| 6 | +Geglu_bfloat16_ND_fuzz_7,aclnnGeGlu,"('bfloat16', 'bfloat16', 'bfloat16')","('ND',)","((4,), (2,), (2,))","((-1, 1),)","(-2,-1)","{'dim': 0, 'approximate': 1}",0.0001 | ||
| @@ -7,20 +7,23 @@ | |||
| 7 | # THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, | 7 | # THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, |
| 8 | # INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. | 8 | # INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. |
| 9 | # See LICENSE in the root of the software repository for the full text of the License. | 9 | # See LICENSE in the root of the software repository for the full text of the License. |
| 10 | -''' | 10 | +""" |
| 11 | gelu golden function | 11 | gelu golden function |
| 12 | -''' | 12 | +""" |
| 13 | + | ||
| 13 | import numpy as np | 14 | import numpy as np |
| 15 | +import torch | ||
| 14 | 16 | ||
| 15 | __golden__ = { | 17 | __golden__ = { |
| 16 | - "kernel": { | 18 | + "aclnn": { |
| 17 | - "gelu": "gelu_golden" | 19 | + "aclnnGelu": "aclnn_gelu_golden", |
| 18 | - } | 20 | + }, |
| 21 | + "kernel": {"gelu": "gelu_golden"}, | ||
| 19 | } | 22 | } |
| 20 | 23 | ||
| 21 | 24 | ||
| 22 | def gelu_golden(x, approximate="tanh", **kwargs): | 25 | def gelu_golden(x, approximate="tanh", **kwargs): |
| 23 | - ''' | 26 | + """ |
| 24 | Golden function for gelu. | 27 | Golden function for gelu. |
| 25 | All the parameters (names and order) follow @gelu_def.cpp without outputs. | 28 | All the parameters (names and order) follow @gelu_def.cpp without outputs. |
| 26 | All the input Tensors are numpy.ndarray. | 29 | All the input Tensors are numpy.ndarray. |
| @@ -31,16 +34,26 @@ def gelu_golden(x, approximate="tanh", **kwargs): | |||
| 31 | 34 | ||
| 32 | Returns: | 35 | Returns: |
| 33 | Output tensor | 36 | Output tensor |
| 34 | - ''' | 37 | + """ |
| 35 | import torch | 38 | import torch |
| 36 | - | 39 | + |
| 37 | input_dtype = x.dtype | 40 | input_dtype = x.dtype |
| 38 | - | 41 | + |
| 39 | # Promote float16 and bfloat16 to float32 for computation | 42 | # Promote float16 and bfloat16 to float32 for computation |
| 40 | if input_dtype.name == "float16" or input_dtype.name == "bfloat16": | 43 | if input_dtype.name == "float16" or input_dtype.name == "bfloat16": |
| 41 | x = x.astype(np.float32) | 44 | x = x.astype(np.float32) |
| 42 | - | 45 | + |
| 43 | m = torch.nn.GELU(approximate=approximate) | 46 | m = torch.nn.GELU(approximate=approximate) |
| 44 | res = m(torch.from_numpy(x)).numpy() | 47 | res = m(torch.from_numpy(x)).numpy() |
| 45 | - | 48 | + |
| 46 | return res.astype(input_dtype, copy=False) | 49 | return res.astype(input_dtype, copy=False) |
| 50 | + | ||
| 51 | + | ||
| 52 | +def aclnn_gelu_golden(self, out, **kwargs): | ||
| 53 | + orig_dtype = self.dtype | ||
| 54 | + x = self | ||
| 55 | + if x.dtype != torch.float64: | ||
| 56 | + x = x.to(torch.float32) | ||
| 57 | + m = torch.nn.GELU(approximate="tanh") | ||
| 58 | + result = m(x).to(orig_dtype) | ||
| 59 | + return [result] | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,absolute_precision | ||
| 2 | +Gelu_float32_ND_fuzz_1,aclnnGelu,"('float32', 'float32')","('ND',)","((196, 2, 1, 76),(196, 2, 1, 76))","((-1000, -10),)","(-1,)","('float32',)",0.0001 | ||
| 3 | +Gelu_float16_ND_fuzz_2,aclnnGelu,"('float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('float16',)",0.001 | ||
| 4 | +Gelu_float32_ND_fuzz_3,aclnnGelu,"('float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)","('float32',)",0.0001 | ||
| 5 | +Gelu_float16_ND_fuzz_4,aclnnGelu,"('float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('float16',)",0.001 | ||
| 6 | +Gelu_float16_ND_fuzz_5,aclnnGelu,"('float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)","('float16',)",0.001 | ||
| @@ -13,9 +13,10 @@ import numpy as np | |||
| 13 | 13 | ||
| 14 | 14 | ||
| 15 | __golden__ = { | 15 | __golden__ = { |
| 16 | - "kernel": { | 16 | + "aclnn": { |
| 17 | - "gelu_grad": "gelu_grad_golden" | 17 | + "aclnnGeluBackward": "aclnn_gelu_backward_golden", |
| 18 | - } | 18 | + }, |
| 19 | + "kernel": {"gelu_grad": "gelu_grad_golden"}, | ||
| 19 | } | 20 | } |
| 20 | 21 | ||
| 21 | _MIN_FP32 = np.float32(2 ** (-126)) | 22 | _MIN_FP32 = np.float32(2 ** (-126)) |
| @@ -83,7 +84,7 @@ def _result_grad_compute(data_x): | |||
| 83 | 84 | ||
| 84 | 85 | ||
| 85 | def gelu_grad_golden(dy, x, y, **kwargs): | 86 | def gelu_grad_golden(dy, x, y, **kwargs): |
| 86 | - ''' | 87 | + """ |
| 87 | Golden function for gelu_grad. | 88 | Golden function for gelu_grad. |
| 88 | All the parameters (names and order) follow @gelu_grad_def.cpp without outputs. | 89 | All the parameters (names and order) follow @gelu_grad_def.cpp without outputs. |
| 89 | All the input Tensors are numpy.ndarray. | 90 | All the input Tensors are numpy.ndarray. |
| @@ -94,19 +95,19 @@ def gelu_grad_golden(dy, x, y, **kwargs): | |||
| 94 | 95 | ||
| 95 | Returns: | 96 | Returns: |
| 96 | Output tensor | 97 | Output tensor |
| 97 | - ''' | 98 | + """ |
| 98 | import torch | 99 | import torch |
| 99 | from packaging import version | 100 | from packaging import version |
| 100 | 101 | ||
| 101 | input_dtype = dy.dtype | 102 | input_dtype = dy.dtype |
| 102 | has_improve_precision = False | 103 | has_improve_precision = False |
| 103 | - | 104 | + |
| 104 | if version.parse(torch.__version__) >= version.parse("1.12.0"): | 105 | if version.parse(torch.__version__) >= version.parse("1.12.0"): |
| 105 | if input_dtype.name in ["float16", "bfloat16"]: | 106 | if input_dtype.name in ["float16", "bfloat16"]: |
| 106 | dy = dy.astype(np.float32) | 107 | dy = dy.astype(np.float32) |
| 107 | x = x.astype(np.float32) | 108 | x = x.astype(np.float32) |
| 108 | y = y.astype(np.float32) | 109 | y = y.astype(np.float32) |
| 109 | - | 110 | + |
| 110 | dy_torch = torch.from_numpy(dy) | 111 | dy_torch = torch.from_numpy(dy) |
| 111 | x_torch = torch.from_numpy(x) | 112 | x_torch = torch.from_numpy(x) |
| 112 | result = torch.ops.aten.gelu_backward(dy_torch, x_torch, approximate="tanh") | 113 | result = torch.ops.aten.gelu_backward(dy_torch, x_torch, approximate="tanh") |
| @@ -127,3 +128,19 @@ def gelu_grad_golden(dy, x, y, **kwargs): | |||
| 127 | if has_improve_precision: | 128 | if has_improve_precision: |
| 128 | result = result.astype(input_dtype, copy=False) | 129 | result = result.astype(input_dtype, copy=False) |
| 129 | return result | 130 | return result |
| 131 | + | ||
| 132 | + | ||
| 133 | +def aclnn_gelu_backward_golden(gradOutput, self, gradInput, **kwargs): | ||
| 134 | + """ | ||
| 135 | + Aclnn golden for aclnnGeluBackward. | ||
| 136 | + Parameters follow @aclnnGeluBackwardGetWorkspaceSize without workspaceSize & executor. | ||
| 137 | + All the input Tensors are torch.Tensor. | ||
| 138 | + """ | ||
| 139 | + import torch | ||
| 140 | + | ||
| 141 | + input_dy = gradOutput | ||
| 142 | + input_x = self | ||
| 143 | + | ||
| 144 | + result = torch.ops.aten.gelu_backward(input_dy, input_x, approximate="tanh") | ||
| 145 | + | ||
| 146 | + return result | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,absolute_precision | ||
| 2 | +GeluGrad_float32_ND_fuzz_1,aclnnGeluBackward,"('float32', 'float32', 'float32')","('ND',)","((196, 2, 1, 76),(196, 2, 1, 76),(196, 2, 1, 76))","((-1000, -10),)","(-1,)","('float32',)",0.0001 | ||
| 3 | +GeluGrad_float16_ND_fuzz_2,aclnnGeluBackward,"('float16', 'float16','float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('float16',)",0.001 | ||
| 4 | +GeluGrad_float32_ND_fuzz_3,aclnnGeluBackward,"('float32', 'float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17),(1, 1, 4, 11, 48, 17))","((-10, -2),)","(-1,)","('float32',)",0.0001 | ||
| 5 | +GeluGrad_float16_ND_fuzz_4,aclnnGeluBackward,"('float16', 'float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192),(139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('float16',)",0.001 | ||
| 6 | +GeluGrad_float16_ND_fuzz_5,aclnnGeluBackward,"('float16', 'float16','float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1),(1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)","('float16',)",0.001 | ||
| @@ -12,7 +12,12 @@ | |||
| 12 | 12 | ||
| 13 | import numpy as np | 13 | import numpy as np |
| 14 | 14 | ||
| 15 | -__golden__ = {"kernel": {"hardtanh_grad": "hardtanh_grad_golden"}} | 15 | +__golden__ = { |
| 16 | + "aclnn": { | ||
| 17 | + "aclnnHardtanhBackward": "aclnn_hardtanh_backward_golden", | ||
| 18 | + }, | ||
| 19 | + "kernel": {"hardtanh_grad": "hardtanh_grad_golden"}, | ||
| 20 | +} | ||
| 16 | 21 | ||
| 17 | 22 | ||
| 18 | def hardtanh_grad_golden(result, grad, *, min_val=-1.0, max_val=1.0, **kwargs): | 23 | def hardtanh_grad_golden(result, grad, *, min_val=-1.0, max_val=1.0, **kwargs): |
| @@ -38,3 +43,12 @@ def hardtanh_grad_golden(result, grad, *, min_val=-1.0, max_val=1.0, **kwargs): | |||
| 38 | if str(in_data_type) == "float16" or str(in_data_type) == "bfloat16": | 43 | if str(in_data_type) == "float16" or str(in_data_type) == "bfloat16": |
| 39 | res = res.astype(in_data_type, copy=False) | 44 | res = res.astype(in_data_type, copy=False) |
| 40 | return res | 45 | return res |
| 46 | + | ||
| 47 | + | ||
| 48 | +def aclnn_hardtanh_backward_golden(gradOutput, self, min, max, out, **kwargs): | ||
| 49 | + if hasattr(min, "item"): | ||
| 50 | + min = min.item() | ||
| 51 | + if hasattr(max, "item"): | ||
| 52 | + max = max.item() | ||
| 53 | + mask = (self > min) & (self < max) | ||
| 54 | + return [gradOutput * mask.to(gradOutput.dtype)] | ||
| @@ -0,0 +1,4 @@ | |||
| 1 | +testcase_name,api_name,tensor_view_shapes,tensor_dtypes,scalar_dtypes,scalar_data_ranges | ||
| 2 | +aclnnHardtanhBackward_1,aclnnHardtanhBackward,"((1,2,3),(1,2,3),(1,2,3))","('bfloat16','bfloat16','bfloat16')","('float16','float16')","((-1, -1),(2, 2))" | ||
| 3 | +aclnnHardtanhBackward_2,aclnnHardtanhBackward,"((1,2,3),(1,2,3),(1,2,3))","('float16','float16','float16')","('float16','float16')","((-2, -2),(3, 3))" | ||
| 4 | +aclnnHardtanhBackward_3,aclnnHardtanhBackward,"((1,2,3),(1,2,3),(1,2,3))","('float32','float32','float32')","('float16','float16')","((10, 10),(20, 20))" | ||
| @@ -11,11 +11,17 @@ | |||
| 11 | 11 | ||
| 12 | import numpy as np | 12 | import numpy as np |
| 13 | 13 | ||
| 14 | -__golden__ = {"kernel": {"leaky_relu": "leaky_relu_golden"}} | 14 | +__golden__ = { |
| 15 | + "aclnn": { | ||
| 16 | + "aclnnLeakyRelu": "aclnn_leaky_relu_golden", | ||
| 17 | + "aclnnInplaceLeakyRelu": "aclnn_inplace_leaky_relu_golden", | ||
| 18 | + }, | ||
| 19 | + "kernel": {"leaky_relu": "leaky_relu_golden"}, | ||
| 20 | +} | ||
| 15 | 21 | ||
| 16 | 22 | ||
| 17 | def leaky_relu_golden(x, *, negative_slope=0, **kwargs): | 23 | def leaky_relu_golden(x, *, negative_slope=0, **kwargs): |
| 18 | - ''' | 24 | + """ |
| 19 | Golden function for leaky_relu. | 25 | Golden function for leaky_relu. |
| 20 | All the parameters (names and order) follow @leaky_relu_def.cpp without outputs. | 26 | All the parameters (names and order) follow @leaky_relu_def.cpp without outputs. |
| 21 | All the input Tensors are numpy.ndarray. | 27 | All the input Tensors are numpy.ndarray. |
| @@ -28,7 +34,7 @@ def leaky_relu_golden(x, *, negative_slope=0, **kwargs): | |||
| 28 | 34 | ||
| 29 | Returns: | 35 | Returns: |
| 30 | Output tensor | 36 | Output tensor |
| 31 | - ''' | 37 | + """ |
| 32 | import torch | 38 | import torch |
| 33 | 39 | ||
| 34 | if "bfloat16" in x.dtype.name: | 40 | if "bfloat16" in x.dtype.name: |
| @@ -40,3 +46,29 @@ def leaky_relu_golden(x, *, negative_slope=0, **kwargs): | |||
| 40 | if "bfloat16" in x.dtype.name: | 46 | if "bfloat16" in x.dtype.name: |
| 41 | return result.view(torch.int16).numpy().view(x.dtype) | 47 | return result.view(torch.int16).numpy().view(x.dtype) |
| 42 | return result.numpy() | 48 | return result.numpy() |
| 49 | + | ||
| 50 | + | ||
| 51 | +def aclnn_inplace_leaky_relu_golden(selfRef, negativeSlope, **kwargs): | ||
| 52 | + """ | ||
| 53 | + Aclnn golden for aclnnInplaceLeakyRelu. | ||
| 54 | + Parameters follow @aclnnInplaceLeakyReluGetWorkspaceSize without workspaceSize & executor. | ||
| 55 | + All the input Tensors are torch.Tensor. | ||
| 56 | + """ | ||
| 57 | + import torch | ||
| 58 | + | ||
| 59 | + if hasattr(negativeSlope, "item"): | ||
| 60 | + negativeSlope = negativeSlope.item() | ||
| 61 | + return [torch.nn.functional.leaky_relu(selfRef, negative_slope=negativeSlope)] | ||
| 62 | + | ||
| 63 | + | ||
| 64 | +def aclnn_leaky_relu_golden(self, negativeSlope, out=None, **kwargs): | ||
| 65 | + """ | ||
| 66 | + Aclnn golden for aclnnLeakyRelu. | ||
| 67 | + Parameters follow @aclnnLeakyReluGetWorkspaceSize without workspaceSize & executor. | ||
| 68 | + All the input Tensors are torch.Tensor. | ||
| 69 | + """ | ||
| 70 | + import torch | ||
| 71 | + | ||
| 72 | + if hasattr(negativeSlope, "item"): | ||
| 73 | + negativeSlope = negativeSlope.item() | ||
| 74 | + return [torch.nn.functional.leaky_relu(self, negative_slope=negativeSlope)] | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,scalar_dtypes,scalar_data_ranges,attributes,absolute_precision | ||
| 2 | +InplaceLeakyRelu_float32_ND_fuzz_1,aclnnInplaceLeakyRelu,('float32'),"('ND',)","((196, 2, 1, 76, 1, 1))","((-1000, -10),)","(-1,)","('float32',)",('float16'),"(0, 1)",{'negative_slope': 0.0006944149602797177},0.0001 | ||
| 3 | +InplaceLeakyRelu_float16_ND_fuzz_2,aclnnInplaceLeakyRelu,('float16'),"('ND',)","((192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('float16',)",('float16'),"(0, 1)",{'negative_slope': 0.007353729707176569},0.001 | ||
| 4 | +InplaceLeakyRelu_float32_ND_fuzz_3,aclnnInplaceLeakyRelu,('float32'),"('ND',)","((1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)","('float32',)",('float16'),"(0, 1)",{'negative_slope': 0.003578863762855549},0.0001 | ||
| 5 | +InplaceLeakyRelu_float16_ND_fuzz_4,aclnnInplaceLeakyRelu,('float16'),"('ND',)","((139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('float16',)",('float16'),"(0, 1)",{'negative_slope': 0.001118463241933789},0.001 | ||
| 6 | +InplaceLeakyRelu_float16_ND_fuzz_5,aclnnInplaceLeakyRelu,('float16'),"('ND',)","((1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)","('float16',)",('float16'),"(0, 1)",{'negative_slope': 0.006011051318991595},0.001 | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,scalar_dtypes,scalar_data_ranges,absolute_precision | ||
| 2 | +LeakyRelu_float32_ND_fuzz_1,aclnnLeakyRelu,"('float32', 'float32')","('ND',)","((196, 2, 1, 76, 1, 1), (196, 2, 1, 76, 1, 1))","((-1000, -10),)","(-1,)","('float32',)",('float16'),"(0, 1)",0.0001 | ||
| 3 | +LeakyRelu_float16_ND_fuzz_2,aclnnLeakyRelu,"('float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('float16',)",('float16'),"(0, 1)",0.001 | ||
| 4 | +LeakyRelu_float32_ND_fuzz_3,aclnnLeakyRelu,"('float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)","('float32',)",('float16'),"(0, 1)",0.0001 | ||
| 5 | +LeakyRelu_float16_ND_fuzz_4,aclnnLeakyRelu,"('float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('float16',)",('float16'),"(0, 1)",0.001 | ||
| 6 | +LeakyRelu_float16_ND_fuzz_5,aclnnLeakyRelu,"('float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)","('float16',)",('float16'),"(0, 1)",0.001 | ||
| @@ -11,12 +11,18 @@ | |||
| 11 | # ---------------------------------------------------------------------------- | 11 | # ---------------------------------------------------------------------------- |
| 12 | 12 | ||
| 13 | import numpy as np | 13 | import numpy as np |
| 14 | +import torch | ||
| 14 | 15 | ||
| 15 | -__golden__ = {"kernel": {"relu": "relu_golden"}} | 16 | +__golden__ = { |
| 17 | + "aclnn": { | ||
| 18 | + "aclnnRelu": "aclnn_relu_golden", | ||
| 19 | + }, | ||
| 20 | + "kernel": {"relu": "relu_golden"}, | ||
| 21 | +} | ||
| 16 | 22 | ||
| 17 | 23 | ||
| 18 | def relu_golden(x, **kwargs): | 24 | def relu_golden(x, **kwargs): |
| 19 | - ''' | 25 | + """ |
| 20 | Golden function for relu. | 26 | Golden function for relu. |
| 21 | All the parameters (names and order) follow @relu_def.cpp without outputs. | 27 | All the parameters (names and order) follow @relu_def.cpp without outputs. |
| 22 | All the input Tensors are numpy.ndarray. | 28 | All the input Tensors are numpy.ndarray. |
| @@ -27,7 +33,16 @@ def relu_golden(x, **kwargs): | |||
| 27 | 33 | ||
| 28 | Returns: | 34 | Returns: |
| 29 | Output tensor | 35 | Output tensor |
| 30 | - ''' | 36 | + """ |
| 31 | data_res = np.maximum(x, 0) | 37 | data_res = np.maximum(x, 0) |
| 32 | data_res[np.logical_and(data_res == 0, np.signbit(data_res))] = 0.0 | 38 | data_res[np.logical_and(data_res == 0, np.signbit(data_res))] = 0.0 |
| 33 | return data_res | 39 | return data_res |
| 40 | + | ||
| 41 | + | ||
| 42 | +def aclnn_relu_golden(self, out=None, **kwargs): | ||
| 43 | + """ | ||
| 44 | + Aclnn golden for aclnnRelu. | ||
| 45 | + Parameters follow @aclnnReluGetWorkspaceSize without workspaceSize & executor. | ||
| 46 | + All the input Tensors are torch.Tensor. | ||
| 47 | + """ | ||
| 48 | + return [torch.relu(self)] | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,absolute_precision | ||
| 2 | +Relu_float32_ND_fuzz_1,aclnnRelu,"('float32', 'float32')","('ND',)","((196, 2, 1, 76, 1, 1), (196, 2, 1, 76, 1, 1))","((-1000, -10),)","(-1,)","('float32',)",0.0001 | ||
| 3 | +Relu_float16_ND_fuzz_2,aclnnRelu,"('float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)","('float16',)",0.001 | ||
| 4 | +Relu_float32_ND_fuzz_3,aclnnRelu,"('float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)","('float32',)",0.0001 | ||
| 5 | +Relu_float16_ND_fuzz_4,aclnnRelu,"('float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)","('float16',)",0.001 | ||
| 6 | +Relu_float16_ND_fuzz_5,aclnnRelu,"('float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)","('float16',)",0.001 | ||
| @@ -11,12 +11,19 @@ | |||
| 11 | # ---------------------------------------------------------------------------- | 11 | # ---------------------------------------------------------------------------- |
| 12 | 12 | ||
| 13 | import numpy as np | 13 | import numpy as np |
| 14 | +import torch | ||
| 14 | 15 | ||
| 15 | -__golden__ = {"kernel": {"sigmoid": "sigmoid_golden"}} | 16 | +__golden__ = { |
| 17 | + "aclnn": { | ||
| 18 | + "aclnnSigmoid": "aclnn_sigmoid_golden", | ||
| 19 | + "aclnnInplaceSigmoid": "aclnn_inplace_sigmoid_golden", | ||
| 20 | + }, | ||
| 21 | + "kernel": {"sigmoid": "sigmoid_golden"}, | ||
| 22 | +} | ||
| 16 | 23 | ||
| 17 | 24 | ||
| 18 | def sigmoid_golden(x, **kwargs): | 25 | def sigmoid_golden(x, **kwargs): |
| 19 | - ''' | 26 | + """ |
| 20 | Golden function for sigmoid. | 27 | Golden function for sigmoid. |
| 21 | All the parameters (names and order) follow @sigmoid_def.cpp without outputs. | 28 | All the parameters (names and order) follow @sigmoid_def.cpp without outputs. |
| 22 | All the input Tensors are numpy.ndarray. | 29 | All the input Tensors are numpy.ndarray. |
| @@ -27,7 +34,7 @@ def sigmoid_golden(x, **kwargs): | |||
| 27 | 34 | ||
| 28 | Returns: | 35 | Returns: |
| 29 | Output tensor | 36 | Output tensor |
| 30 | - ''' | 37 | + """ |
| 31 | input_dtype = x.dtype | 38 | input_dtype = x.dtype |
| 32 | if input_dtype.name in ("bfloat16",): | 39 | if input_dtype.name in ("bfloat16",): |
| 33 | x = x.astype("float32") | 40 | x = x.astype("float32") |
| @@ -36,3 +43,25 @@ def sigmoid_golden(x, **kwargs): | |||
| 36 | tensor_add = tensor_exp + 1 | 43 | tensor_add = tensor_exp + 1 |
| 37 | res = 1 / tensor_add | 44 | res = 1 / tensor_add |
| 38 | return res.astype(input_dtype, copy=False) | 45 | return res.astype(input_dtype, copy=False) |
| 46 | + | ||
| 47 | + | ||
| 48 | +def aclnn_inplace_sigmoid_golden(selfRef=None, **kwargs): | ||
| 49 | + """ | ||
| 50 | + Aclnn golden for aclnnInplaceSigmoid. | ||
| 51 | + Parameters follow @aclnnInplaceSigmoidGetWorkspaceSize without workspaceSize & executor. | ||
| 52 | + All the input Tensors are torch.Tensor. | ||
| 53 | + """ | ||
| 54 | + return [ | ||
| 55 | + torch.nn.functional.sigmoid(selfRef) | ||
| 56 | + if hasattr(torch.nn.functional, "sigmoid") | ||
| 57 | + else torch.sigmoid(selfRef) | ||
| 58 | + ] | ||
| 59 | + | ||
| 60 | + | ||
| 61 | +def aclnn_sigmoid_golden(self, out=None, **kwargs): | ||
| 62 | + """ | ||
| 63 | + Aclnn golden for aclnnSigmoid. | ||
| 64 | + Parameters follow @aclnnSigmoidGetWorkspaceSize without workspaceSize & executor. | ||
| 65 | + All the input Tensors are torch.Tensor. | ||
| 66 | + """ | ||
| 67 | + return [torch.sigmoid(self)] | ||
| @@ -0,0 +1,4 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,absolute_precision | ||
| 2 | +aclnn_sigmoid_fuzz_1,aclnnSigmoid,"('float32', 'float32')","('ND',)","((19, 2, 5, 8), (19, 2, 5, 8))","((-1000, 1000),)","(-1,)","('float32',)",0.0001 | ||
| 3 | +aclnn_sigmoid_fuzz_2,aclnnSigmoid,"('float16', 'float16')","('ND',)","((19, 2, 5, 8), (19, 2, 5, 8))","((-1000, 1000),)","(-1,)","('float16',)",0.0001 | ||
| 4 | +aclnn_sigmoid_fuzz_6,aclnnSigmoid,"('bfloat16', 'bfloat16')","('ND',)","((19, 2, 5, 8), (19, 2, 5, 8))","((-1000, 1000),)","(-1,)","('bfloat16',)",0.0001 | ||
| @@ -11,12 +11,18 @@ | |||
| 11 | # ---------------------------------------------------------------------------- | 11 | # ---------------------------------------------------------------------------- |
| 12 | 12 | ||
| 13 | import numpy as np | 13 | import numpy as np |
| 14 | +import torch | ||
| 14 | 15 | ||
| 15 | -__golden__ = {"kernel": {"sigmoid_grad": "sigmoid_grad_golden"}} | 16 | +__golden__ = { |
| 17 | + "aclnn": { | ||
| 18 | + "aclnnSigmoidBackward": "aclnn_sigmoid_backward_golden", | ||
| 19 | + }, | ||
| 20 | + "kernel": {"sigmoid_grad": "sigmoid_grad_golden"}, | ||
| 21 | +} | ||
| 16 | 22 | ||
| 17 | 23 | ||
| 18 | def sigmoid_grad_golden(y, dy, **kwargs): | 24 | def sigmoid_grad_golden(y, dy, **kwargs): |
| 19 | - ''' | 25 | + """ |
| 20 | Golden function for sigmoid_grad. | 26 | Golden function for sigmoid_grad. |
| 21 | All the parameters (names and order) follow @sigmoid_grad_def.cpp without outputs. | 27 | All the parameters (names and order) follow @sigmoid_grad_def.cpp without outputs. |
| 22 | All the input Tensors are numpy.ndarray. | 28 | All the input Tensors are numpy.ndarray. |
| @@ -27,15 +33,24 @@ def sigmoid_grad_golden(y, dy, **kwargs): | |||
| 27 | 33 | ||
| 28 | Returns: | 34 | Returns: |
| 29 | Output tensor | 35 | Output tensor |
| 30 | - ''' | 36 | + """ |
| 31 | dtype = y.dtype | 37 | dtype = y.dtype |
| 32 | - | 38 | + |
| 33 | - if 'float16' in str(dtype): | 39 | + if "float16" in str(dtype): |
| 34 | y = y.astype("float32") | 40 | y = y.astype("float32") |
| 35 | dy = dy.astype("float32") | 41 | dy = dy.astype("float32") |
| 36 | - | 42 | + |
| 37 | tensor_sub = np.subtract(1.0, y) | 43 | tensor_sub = np.subtract(1.0, y) |
| 38 | tensor_mul = np.multiply(tensor_sub, dy) | 44 | tensor_mul = np.multiply(tensor_sub, dy) |
| 39 | res = np.multiply(tensor_mul, y) | 45 | res = np.multiply(tensor_mul, y) |
| 40 | - | 46 | + |
| 41 | return res.astype(dtype, copy=False) | 47 | return res.astype(dtype, copy=False) |
| 48 | + | ||
| 49 | + | ||
| 50 | +def aclnn_sigmoid_backward_golden(gradOutput, output, gradInput=None, **kwargs): | ||
| 51 | + orig_dtype = gradOutput.dtype | ||
| 52 | + if orig_dtype in (torch.float16, torch.bfloat16): | ||
| 53 | + gradOutput = gradOutput.to(torch.float32) | ||
| 54 | + output = output.to(torch.float32) | ||
| 55 | + result = gradOutput * output * (1 - output) | ||
| 56 | + return [result.to(orig_dtype)] | ||
| @@ -0,0 +1,4 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,absolute_precision | ||
| 2 | +aclnn_sigmoid_backward_fuzz_1,aclnnSigmoidBackward,"('float32', 'float32', 'float32')","('ND',)","((19, 2, 5, 8), (19, 2, 5, 8), (19, 2, 5, 8))","((-1000, 1000),)","(-1,)","('float32',)",0.0001 | ||
| 3 | +aclnn_sigmoid_backward_fuzz_2,aclnnSigmoidBackward,"('float16', 'float16', 'float16')","('ND',)","((19, 2, 5, 8), (19, 2, 5, 8), (19, 2, 5, 8))","((-1000, 1000),)","(-1,)","('float16',)",0.0001 | ||
| 4 | +aclnn_sigmoid_backward_fuzz_3,aclnnSigmoidBackward,"('bfloat16', 'bfloat16', 'bfloat16')","('ND',)","((19, 2, 5, 8), (19, 2, 5, 8), (19, 2, 5, 8))","((-1000, 1000),)","(-1,)","('bfloat16',)",0.0001 | ||
| @@ -12,11 +12,16 @@ | |||
| 12 | 12 | ||
| 13 | import numpy as np | 13 | import numpy as np |
| 14 | 14 | ||
| 15 | -__golden__ = {"kernel": {"silu_grad": "silu_grad_golden"}} | 15 | +__golden__ = { |
| 16 | + "aclnn": { | ||
| 17 | + "aclnnSiluBackward": "aclnn_silu_backward_golden", | ||
| 18 | + }, | ||
| 19 | + "kernel": {"silu_grad": "silu_grad_golden"}, | ||
| 20 | +} | ||
| 16 | 21 | ||
| 17 | 22 | ||
| 18 | def silu_grad_golden(dy, x, **kwargs): | 23 | def silu_grad_golden(dy, x, **kwargs): |
| 19 | - ''' | 24 | + """ |
| 20 | Golden function for silu_grad. | 25 | Golden function for silu_grad. |
| 21 | All the parameters (names and order) follow @silu_grad_def.cpp without outputs. | 26 | All the parameters (names and order) follow @silu_grad_def.cpp without outputs. |
| 22 | All the input Tensors are numpy.ndarray. | 27 | All the input Tensors are numpy.ndarray. |
| @@ -27,11 +32,11 @@ def silu_grad_golden(dy, x, **kwargs): | |||
| 27 | 32 | ||
| 28 | Returns: | 33 | Returns: |
| 29 | Output tensor | 34 | Output tensor |
| 30 | - ''' | 35 | + """ |
| 31 | import torch | 36 | import torch |
| 32 | 37 | ||
| 33 | def _to_torch(arr): | 38 | def _to_torch(arr): |
| 34 | - if arr.dtype.name == 'bfloat16': | 39 | + if arr.dtype.name == "bfloat16": |
| 35 | return torch.from_numpy(arr.view(np.int16)).view(torch.bfloat16) | 40 | return torch.from_numpy(arr.view(np.int16)).view(torch.bfloat16) |
| 36 | return torch.from_numpy(arr) | 41 | return torch.from_numpy(arr) |
| 37 | 42 | ||
| @@ -40,6 +45,17 @@ def silu_grad_golden(dy, x, **kwargs): | |||
| 40 | dx = torch.ops.aten.silu_backward(dy_torch, x_torch) | 45 | dx = torch.ops.aten.silu_backward(dy_torch, x_torch) |
| 41 | 46 | ||
| 42 | if dx.dtype == torch.bfloat16: | 47 | if dx.dtype == torch.bfloat16: |
| 43 | - return dx.view(torch.int16).numpy().view(np.dtype('bfloat16')) | 48 | + return dx.view(torch.int16).numpy().view(np.dtype("bfloat16")) |
| 44 | else: | 49 | else: |
| 45 | return dx.numpy() | 50 | return dx.numpy() |
| 51 | + | ||
| 52 | + | ||
| 53 | +def aclnn_silu_backward_golden(gradOutput, self, gradInput=None, **kwargs): | ||
| 54 | + """ | ||
| 55 | + Aclnn golden for aclnnSiluBackward. | ||
| 56 | + Parameters follow @aclnnSiluBackwardGetWorkspaceSize without workspaceSize & executor. | ||
| 57 | + All the input Tensors are torch.Tensor. | ||
| 58 | + """ | ||
| 59 | + import torch | ||
| 60 | + | ||
| 61 | + return [torch.ops.aten.silu_backward(gradOutput, self)] | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,absolute_precision | ||
| 2 | +SiluBackward_float32_ND_fuzz_1,aclnnSiluBackward,"('float32', 'float32', 'float32')","('ND',)","((196, 2, 1, 76, 1, 1), (196, 2, 1, 76, 1, 1), (196, 2, 1, 76, 1, 1))","((-1000, -10),)","(-1,)",0.0001 | ||
| 3 | +SiluBackward_float16_ND_fuzz_2,aclnnSiluBackward,"('float16', 'float16', 'float16')","('ND',)","((192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1), (192, 64, 1, 1, 1, 5, 1))","((0, 0),)","(-1,)",0.001 | ||
| 4 | +SiluBackward_float32_ND_fuzz_3,aclnnSiluBackward,"('float32', 'float32', 'float32')","('ND',)","((1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17), (1, 1, 4, 11, 48, 17))","((-3.4e+38, 3.4e+38),)","(-1,)",0.0001 | ||
| 5 | +SiluBackward_float16_ND_fuzz_4,aclnnSiluBackward,"('float16', 'float16', 'float16')","('ND',)","((139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192), (139, 3, 1, 1, 1, 192))","((-10, -2),)","(-1,)",0.001 | ||
| 6 | +SiluBackward_float16_ND_fuzz_5,aclnnSiluBackward,"('float16', 'float16', 'float16')","('ND',)","((1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1), (1, 139, 1, 28, 1, 22, 1))","((-1, 1),)","(-1,)",0.001 | ||
| @@ -11,12 +11,18 @@ | |||
| 11 | # ---------------------------------------------------------------------------- | 11 | # ---------------------------------------------------------------------------- |
| 12 | 12 | ||
| 13 | import numpy as np | 13 | import numpy as np |
| 14 | +import torch | ||
| 14 | 15 | ||
| 15 | -__golden__ = {"kernel": {"swish": "swish_golden"}} | 16 | +__golden__ = { |
| 17 | + "aclnn": { | ||
| 18 | + "aclnnSwish": "aclnn_swish_golden", | ||
| 19 | + }, | ||
| 20 | + "kernel": {"swish": "swish_golden"}, | ||
| 21 | +} | ||
| 16 | 22 | ||
| 17 | 23 | ||
| 18 | def swish_golden(x, *, scale=1.0, **kwargs): | 24 | def swish_golden(x, *, scale=1.0, **kwargs): |
| 19 | - ''' | 25 | + """ |
| 20 | Golden function for swish. | 26 | Golden function for swish. |
| 21 | All the parameters (names and order) follow @swish_def.cpp without outputs. | 27 | All the parameters (names and order) follow @swish_def.cpp without outputs. |
| 22 | All the input Tensors are numpy.ndarray. | 28 | All the input Tensors are numpy.ndarray. |
| @@ -27,13 +33,13 @@ def swish_golden(x, *, scale=1.0, **kwargs): | |||
| 27 | 33 | ||
| 28 | Returns: | 34 | Returns: |
| 29 | Output tensor | 35 | Output tensor |
| 30 | - ''' | 36 | + """ |
| 31 | import torch | 37 | import torch |
| 32 | - | 38 | + |
| 33 | dtype = x.dtype | 39 | dtype = x.dtype |
| 34 | - if dtype.name in ('float16', 'bfloat16'): | 40 | + if dtype.name in ("float16", "bfloat16"): |
| 35 | x = x.astype(np.float32) | 41 | x = x.astype(np.float32) |
| 36 | - | 42 | + |
| 37 | if scale == 1.0: | 43 | if scale == 1.0: |
| 38 | x_torch = torch.from_numpy(x) | 44 | x_torch = torch.from_numpy(x) |
| 39 | m = torch.nn.SiLU() | 45 | m = torch.nn.SiLU() |
| @@ -44,11 +50,11 @@ def swish_golden(x, *, scale=1.0, **kwargs): | |||
| 44 | 50 | ||
| 45 | 51 | ||
| 46 | def _swish_overflow(data_input, scale, dtype, **kwargs): | 52 | def _swish_overflow(data_input, scale, dtype, **kwargs): |
| 47 | - short_soc_version = kwargs.get('short_soc_version', '') | 53 | + short_soc_version = kwargs.get("short_soc_version", "") |
| 48 | - | 54 | + |
| 49 | - if dtype.name in ('float16', 'bfloat16'): | 55 | + if dtype.name in ("float16", "bfloat16"): |
| 50 | data_input = data_input.astype(np.float32) | 56 | data_input = data_input.astype(np.float32) |
| 51 | - | 57 | + |
| 52 | if short_soc_version in ("Ascend950",): | 58 | if short_soc_version in ("Ascend950",): |
| 53 | scale_arr = np.array([scale], dtype=data_input.dtype) | 59 | scale_arr = np.array([scale], dtype=data_input.dtype) |
| 54 | multi = data_input * scale_arr * -1.0 | 60 | multi = data_input * scale_arr * -1.0 |
| @@ -59,15 +65,25 @@ def _swish_overflow(data_input, scale, dtype, **kwargs): | |||
| 59 | scale_arr = np.array([scale], dtype=data_input.dtype) | 65 | scale_arr = np.array([scale], dtype=data_input.dtype) |
| 60 | scale_input = np.multiply(data_input, scale_arr) | 66 | scale_input = np.multiply(data_input, scale_arr) |
| 61 | abs_scale_input = np.abs(scale_input) | 67 | abs_scale_input = np.abs(scale_input) |
| 62 | - minus_abs = np.multiply(abs_scale_input, np.array([-1.0], dtype=data_input.dtype)) | 68 | + minus_abs = np.multiply( |
| 69 | + abs_scale_input, np.array([-1.0], dtype=data_input.dtype) | ||
| 70 | + ) | ||
| 63 | sign_diff = np.add(scale_input, minus_abs) | 71 | sign_diff = np.add(scale_input, minus_abs) |
| 64 | half_sign_diff = np.multiply(sign_diff, np.array([0.5], dtype=data_input.dtype)) | 72 | half_sign_diff = np.multiply(sign_diff, np.array([0.5], dtype=data_input.dtype)) |
| 65 | - | 73 | + |
| 66 | exp_top = np.exp(half_sign_diff) | 74 | exp_top = np.exp(half_sign_diff) |
| 67 | exp_bottom = np.exp(minus_abs) | 75 | exp_bottom = np.exp(minus_abs) |
| 68 | one_plus_exp = np.add(exp_bottom, np.array([1.0], dtype=data_input.dtype)) | 76 | one_plus_exp = np.add(exp_bottom, np.array([1.0], dtype=data_input.dtype)) |
| 69 | - | 77 | + |
| 70 | input_mul_exp = np.multiply(data_input, exp_top) | 78 | input_mul_exp = np.multiply(data_input, exp_top) |
| 71 | res = np.divide(input_mul_exp, one_plus_exp) | 79 | res = np.divide(input_mul_exp, one_plus_exp) |
| 72 | - | 80 | + |
| 73 | return res.astype(dtype, copy=False) | 81 | return res.astype(dtype, copy=False) |
| 82 | + | ||
| 83 | + | ||
| 84 | +def aclnn_swish_golden(self, betaOptional=0, out=None, **kwargs): | ||
| 85 | + if hasattr(betaOptional, "item"): | ||
| 86 | + betaOptional = betaOptional.item() | ||
| 87 | + if betaOptional == 1.0 or betaOptional == 0: | ||
| 88 | + return [torch.nn.functional.silu(self)] | ||
| 89 | + return [self * torch.sigmoid(self) ** betaOptional] | ||
| @@ -0,0 +1,4 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,scalar_dtypes,scalar_data_ranges,absolute_precision | ||
| 2 | +aclnn_swish_fuzz_1,aclnnSwish,"('float32', 'float32')","('ND',)","((19, 2, 5, 8), (19, 2, 5, 8))","((-1000, 1000),)","(-1,)","('float32',)","('float32',)","(1, 1)",0.0001 | ||
| 3 | +aclnn_swish_fuzz_2,aclnnSwish,"('float16', 'float16')","('ND',)","((19, 2, 5, 8), (19, 2, 5, 8))","((-1000, 1000),)","(-1,)","('float16',)","('float32',)","(1, 1)",0.0001 | ||
| 4 | +aclnn_swish_fuzz_6,aclnnSwish,"('bfloat16', 'bfloat16')","('ND',)","((19, 2, 5, 8), (19, 2, 5, 8))","((-1000, 1000),)","(-1,)","('bfloat16',)","('float32',)","(1, 1)",0.0001 | ||
| @@ -13,7 +13,12 @@ import torch | |||
| 13 | from ttk.utilities.dtypes import numpy_to_torch_tensor, torch_to_numpy_tensor | 13 | from ttk.utilities.dtypes import numpy_to_torch_tensor, torch_to_numpy_tensor |
| 14 | 14 | ||
| 15 | 15 | ||
| 16 | -__golden__ = {"kernel": {"threshold_grad_v2_d": "threshold_grad_v2_d_golden"}} | 16 | +__golden__ = { |
| 17 | + "aclnn": { | ||
| 18 | + "aclnnThresholdBackward": "aclnn_threshold_backward_golden", | ||
| 19 | + }, | ||
| 20 | + "kernel": {"threshold_grad_v2_d": "threshold_grad_v2_d_golden"}, | ||
| 21 | +} | ||
| 17 | 22 | ||
| 18 | 23 | ||
| 19 | def threshold_grad_v2_d_golden(grad_output, self_tensor, *, threshold=1.0, **kwargs): | 24 | def threshold_grad_v2_d_golden(grad_output, self_tensor, *, threshold=1.0, **kwargs): |
| @@ -32,3 +37,10 @@ def threshold_grad_v2_d_golden(grad_output, self_tensor, *, threshold=1.0, **kwa | |||
| 32 | mask = self_t.to(torch.float32) > float(threshold) | 37 | mask = self_t.to(torch.float32) > float(threshold) |
| 33 | result = torch.where(mask, grad_output_t, torch.zeros_like(grad_output_t)) | 38 | result = torch.where(mask, grad_output_t, torch.zeros_like(grad_output_t)) |
| 34 | return torch_to_numpy_tensor(result.cpu()) | 39 | return torch_to_numpy_tensor(result.cpu()) |
| 40 | + | ||
| 41 | + | ||
| 42 | +def aclnn_threshold_backward_golden(gradOutput, self, threshold, out, **kwargs): | ||
| 43 | + if hasattr(threshold, "item"): | ||
| 44 | + threshold = threshold.item() | ||
| 45 | + mask = (self > threshold).to(gradOutput.dtype) | ||
| 46 | + return [gradOutput * mask] | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes,scalar_dtypes,scalar_data_ranges,absolute_precision | ||
| 2 | +aclnn_relugrad_fuzz_1,aclnnThresholdBackward,"('float32', 'float32', 'float32')","('ND',)","((19, 2, 5, 8), (19, 2, 5, 8), (19, 2, 5, 8))","((-1000, 1000),)","(-1,)","('float32',)",('float32'),"(0, 0)",0.0001 | ||
| 3 | +aclnn_relugrad_fuzz_2,aclnnThresholdBackward,"('float16', 'float16', 'float16')","('ND',)","((19, 2, 5, 8), (19, 2, 5, 8), (19, 2, 5, 8))","((-1000, 1000),)","(-1,)","('float16',)",('float32'),"(0, 0)",0.0001 | ||
| 4 | +aclnn_relugrad_fuzz_3,aclnnThresholdBackward,"('int32', 'int32', 'int32')","('ND',)","((19, 2, 5, 8), (19, 2, 5, 8), (19, 2, 5, 8))","((-1000, 1000),)","(-1,)","('int32',)",('float32'),"(0, 0)",0.0001 | ||
| 5 | +aclnn_relugrad_fuzz_4,aclnnThresholdBackward,"('int8', 'int8', 'int8')","('ND',)","((19, 2, 5, 8), (19, 2, 5, 8), (19, 2, 5, 8))","((-1000, 1000),)","(-1,)","('int8',)",('float32'),"(0, 0)",0.0001 | ||
| 6 | +aclnn_relugrad_fuzz_5,aclnnThresholdBackward,"('uint8', 'uint8', 'uint8')","('ND',)","((19, 2, 5, 8), (19, 2, 5, 8), (19, 2, 5, 8))","((-1000, 1000),)","(-1,)","('uint8',)",('float32'),"(0, 0)",0.0001 | ||
| @@ -11,7 +11,14 @@ | |||
| 11 | # ---------------------------------------------------------------------------- | 11 | # ---------------------------------------------------------------------------- |
| 12 | 12 | ||
| 13 | 13 | ||
| 14 | -__golden__ = {"kernel": {"bucketize_v2": "bucketize_v2_golden"}} | 14 | +import torch |
| 15 | + | ||
| 16 | +__golden__ = { | ||
| 17 | + "aclnn": { | ||
| 18 | + "aclnnBucketize": "aclnn_bucketize_golden", | ||
| 19 | + }, | ||
| 20 | + "kernel": {"bucketize_v2": "bucketize_v2_golden"}, | ||
| 21 | +} | ||
| 15 | 22 | ||
| 16 | 23 | ||
| 17 | def bucketize_v2_golden(x, boundaries, out_int32=False, right=False, **kwargs): | 24 | def bucketize_v2_golden(x, boundaries, out_int32=False, right=False, **kwargs): |
| @@ -29,3 +36,15 @@ def bucketize_v2_golden(x, boundaries, out_int32=False, right=False, **kwargs): | |||
| 29 | boundaries_t = torch.from_numpy(boundaries) | 36 | boundaries_t = torch.from_numpy(boundaries) |
| 30 | res = torch.bucketize(data_t, boundaries_t, out_int32=out_int32, right=right) | 37 | res = torch.bucketize(data_t, boundaries_t, out_int32=out_int32, right=right) |
| 31 | return res.numpy() | 38 | return res.numpy() |
| 39 | + | ||
| 40 | + | ||
| 41 | +def aclnn_bucketize_golden(self, boundaries, outInt32=0, right=0, out=None, **kwargs): | ||
| 42 | + if hasattr(outInt32, "item"): | ||
| 43 | + outInt32 = bool(outInt32.item()) | ||
| 44 | + elif isinstance(outInt32, int): | ||
| 45 | + outInt32 = bool(outInt32) | ||
| 46 | + if hasattr(right, "item"): | ||
| 47 | + right = bool(right.item()) | ||
| 48 | + elif isinstance(right, int): | ||
| 49 | + right = bool(right) | ||
| 50 | + return [torch.bucketize(self, boundaries, out_int32=outInt32, right=right)] | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,output_tensor_indexes,input_data_ranges,attributes | ||
| 2 | +aclnnBucketize_int8_bf16_ND_infnan_random_000000,aclnnBucketize,"['int8', 'bf16', 'int64']","('ND',)","[[3, 9, 4, 1, 5, 4, 6, 4], [31628], [3, 9, 4, 1, 5, 4, 6, 4]]","(-1,)","[[inf, inf], [-5375, 67165]]","{'outInt32':False , 'right':False}" | ||
| 3 | +aclnnBucketize_fp32_fp16_ND_infnan_random_000002,aclnnBucketize,"['fp32', 'fp16', 'int32']","('ND',)","[[9875], [13917], [9875]]","(-1,)","[[-inf, -inf], [-68700, 36819]]","{'outInt32':True , 'right':False}" | ||
| 4 | +aclnnBucketize_int8_int64_ND_infnan_random_000003,aclnnBucketize,"['int8', 'int64', 'int32']","('ND',)","[[3, 5, 2, 610, 7, 3], [9262], [3, 5, 2, 610, 7, 3]]","(-1,)","[[nan, nan], [-81938, 50073]]","{'outInt32':True , 'right':False}" | ||
| 5 | +aclnnBucketize_int64_ND_infnan_random_000004,aclnnBucketize,"['int64', 'int64', 'int32']","('ND',)","[[5, 2, 5, 5, 7], [14382], [5, 2, 5, 5, 7]]","(-1,)","[[0, 1], [-92999, 94508]]","{'outInt32':True , 'right':False}" | ||
| 6 | +aclnnBucketize_int8_fp64_ND_infnan_random_000006,aclnnBucketize,"['int8', 'fp64', 'int64']","('ND',)","[[2, 91, 2, 2888, 2], [36828], [2, 91, 2, 2888, 2]]","(-1,)","[[-inf, -inf], [-96569, 84815]]","{'outInt32':False , 'right':True}" | ||
| @@ -11,7 +11,12 @@ | |||
| 11 | # ---------------------------------------------------------------------------- | 11 | # ---------------------------------------------------------------------------- |
| 12 | 12 | ||
| 13 | 13 | ||
| 14 | -__golden__ = {"kernel": {"embedding": "embedding_golden"}} | 14 | +__golden__ = { |
| 15 | + "aclnn": { | ||
| 16 | + "aclnnEmbedding": "aclnn_embedding_golden", | ||
| 17 | + }, | ||
| 18 | + "kernel": {"embedding": "embedding_golden"}, | ||
| 19 | +} | ||
| 15 | 20 | ||
| 16 | 21 | ||
| 17 | def embedding_golden(x, indices, **kwargs): | 22 | def embedding_golden(x, indices, **kwargs): |
| @@ -53,3 +58,14 @@ def embedding_golden(x, indices, **kwargs): | |||
| 53 | ) | 58 | ) |
| 54 | 59 | ||
| 55 | return res | 60 | return res |
| 61 | + | ||
| 62 | + | ||
| 63 | +def aclnn_embedding_golden(weight, indices, out=None, **kwargs): | ||
| 64 | + """ | ||
| 65 | + Aclnn golden for aclnnEmbedding. | ||
| 66 | + Parameters follow @aclnnEmbeddingGetWorkspaceSize without workspaceSize & executor. | ||
| 67 | + All the input Tensors are torch.Tensor. | ||
| 68 | + """ | ||
| 69 | + import torch | ||
| 70 | + | ||
| 71 | + return torch.nn.functional.embedding(indices, weight) | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,network_name,api_name,tensor_view_shapes,tensor_formats,tensor_dtypes,tensor_storage_shapes,tensor_view_offsets,tensor_view_strides,output_tensor_indexes,output_inplace_indexes,attributes,scalar_dtypes,input_data_ranges,precision_tolerances,absolute_precision,scalar_data_ranges,is_enabled | ||
| 2 | +sdxl_aclnn_func_case_212,UNKNOWN,aclnnEmbedding,"((49408, 768), (1, 77), (1, 77, 768))","('ND', 'ND', 'ND')","('bfloat16', 'int64', 'bfloat16')","((37945344,), (77,), (59136,))","(0, 0, 0)","((768, 1), (77, 1), (59136, 768, 1))","(2,)",(),{},(),"((None, None), (0, 49407))",,1.00E-08,"((None, None),)",TRUE | ||
| 3 | +sdxl_aclnn_func_case_215,UNKNOWN,aclnnEmbedding,"((77, 1280), (1, 77), (1, 77, 1280))","('ND', 'ND', 'ND')","('bfloat16', 'int64', 'bfloat16')","((98560,), (77,), (98560,))","(0, 0, 0)","((1280, 1), (77, 1), (98560, 1280, 1))","(2,)",(),{},(),"((None, None), (0, 76))",,1.00E-08,"((None, None),)",TRUE | ||
| 4 | +sdxl_aclnn_func_case_213,UNKNOWN,aclnnEmbedding,"((77, 768), (1, 77), (1, 77, 768))","('ND', 'ND', 'ND')","('bfloat16', 'int64', 'bfloat16')","((59136,), (77,), (59136,))","(0, 0, 0)","((768, 1), (77, 1), (59136, 768, 1))","(2,)",(),{},(),"((None, None), (0, 76))",,1.00E-08,"((None, None),)",TRUE | ||
| 5 | +sdxl_aclnn_func_case_214,UNKNOWN,aclnnEmbedding,"((49408, 1280), (1, 77), (1, 77, 1280))","('ND', 'ND', 'ND')","('bfloat16', 'int64', 'bfloat16')","((63242240,), (77,), (98560,))","(0, 0, 0)","((1280, 1), (77, 1), (98560, 1280, 1))","(2,)",(),{},(),"((None, None), (0, 49407))",,1.00E-08,"((None, None),)",TRUE | ||
| 6 | +aclnn_embedding_0,UNKNOWN,aclnnEmbedding,"((3, 32), (128, 5), (128, 5, 32))","('ND', 'ND', 'ND')","('float16', 'int64', 'float16')","((96,), (640,), (20480,))","(0, 0, 0)","((32, 1), (5, 1), (160, 32, 1))","(2,)",(),{},(),"((None, None), (0, 2))",,1.00E-08,"((None, None),)",TRUE | ||
| @@ -0,0 +1,34 @@ | |||
| 1 | +#!/usr/bin/env python3 | ||
| 2 | +# -*- coding: UTF-8 -*- | ||
| 3 | +# ---------------------------------------------------------------------------- | ||
| 4 | +# Copyright (c) 2026 Huawei Technologies Co., Ltd. | ||
| 5 | +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of | ||
| 6 | +# CANN Open Software License Agreement Version 2.0 (the "License"). | ||
| 7 | +# Please refer to the License for details. You may not use this file except in compliance with the License. | ||
| 8 | +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, | ||
| 9 | +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. | ||
| 10 | +# See LICENSE in the root of the software repository for the full text of the License. | ||
| 11 | +# ---------------------------------------------------------------------------- | ||
| 12 | +import torch | ||
| 13 | + | ||
| 14 | +__golden__ = { | ||
| 15 | + "aclnn": { | ||
| 16 | + "aclnnGather": "aclnn_gather_golden", | ||
| 17 | + } | ||
| 18 | +} | ||
| 19 | + | ||
| 20 | + | ||
| 21 | +def aclnn_gather_golden(self, dim, index, out=None, **kwargs): | ||
| 22 | + """ | ||
| 23 | + Aclnn golden for aclnnGather. | ||
| 24 | + Parameters follow @aclnnGatherGetWorkspaceSize without workspaceSize & executor. | ||
| 25 | + All the input Tensors are torch.Tensor. | ||
| 26 | + """ | ||
| 27 | + | ||
| 28 | + x = self | ||
| 29 | + tensor_x = x | ||
| 30 | + index = index | ||
| 31 | + dim = dim | ||
| 32 | + np_out = torch.gather(input=tensor_x, dim=dim, index=index) | ||
| 33 | + | ||
| 34 | + return np_out | ||
| @@ -0,0 +1,4 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,attributes,tensor_view_shapes,input_data_ranges,output_tensor_indexes,output_dtypes | ||
| 2 | +gather_elements_test_case_0,aclnnGather,"('bfloat16', 'int64', 'bfloat16')","('ND',)",{ 'dim': -4},"((1, 41, 1, 1, 1), (1, 35, 1, 1, 1), (1, 35, 1, 1, 1))","((0, 0.001), (0, 1))","(2,)",('bfloat16') | ||
| 3 | +gather_elements_test_case_1,aclnnGather,"('int32', 'int64', 'int32')","('ND',)",{ 'dim': 0},"((1, 1, 1, 1, 1, 127, 1, 1), (1, 1, 1, 1, 1, 127, 1, 1), (1, 1, 1, 1, 1, 127, 1, 1))","((-1147483648, 2147483647), (0, 0))","(2,)",('int32') | ||
| 4 | +gather_elements_test_case_2,aclnnGather,"('float32', 'int64', 'float32')","('ND',)",{ 'dim': 0},"((1, 1, 11, 1, 1, 1), (1, 1, 11, 1, 1, 1), (1, 1, 11, 1, 1, 1))","((2, 10), (0, 0))","(2,)",('float32') | ||
| @@ -10,17 +10,20 @@ | |||
| 10 | 10 | ||
| 11 | 11 | ||
| 12 | import numpy as np | 12 | import numpy as np |
| 13 | +import torch | ||
| 13 | 14 | ||
| 14 | 15 | ||
| 15 | __golden__ = { | 16 | __golden__ = { |
| 16 | - "kernel": { | 17 | + "aclnn": { |
| 17 | - "index_fill_d": "index_fill_d_golden" | 18 | + "aclnnInplaceIndexFillTensor": "aclnn_inplace_index_fill_tensor_golden", |
| 18 | - } | 19 | + "aclnnIndexFillTensor": "aclnn_index_fill_tensor_golden", |
| 20 | + }, | ||
| 21 | + "kernel": {"index_fill_d": "index_fill_d_golden"}, | ||
| 19 | } | 22 | } |
| 20 | 23 | ||
| 21 | 24 | ||
| 22 | def index_fill_d_golden(x, assist1, assist2, **kwargs): | 25 | def index_fill_d_golden(x, assist1, assist2, **kwargs): |
| 23 | - ''' | 26 | + """ |
| 24 | Golden function for index_fill_d. | 27 | Golden function for index_fill_d. |
| 25 | All the parameters (names and order) follow @index_fill_d_def.cpp without outputs. | 28 | All the parameters (names and order) follow @index_fill_d_def.cpp without outputs. |
| 26 | All the input Tensors are numpy.ndarray. | 29 | All the input Tensors are numpy.ndarray. |
| @@ -31,6 +34,30 @@ def index_fill_d_golden(x, assist1, assist2, **kwargs): | |||
| 31 | 34 | ||
| 32 | Returns: | 35 | Returns: |
| 33 | Output tensor | 36 | Output tensor |
| 34 | - ''' | 37 | + """ |
| 35 | output_y = np.where(assist1 > 0, x, assist2) | 38 | output_y = np.where(assist1 > 0, x, assist2) |
| 36 | return output_y | 39 | return output_y |
| 40 | + | ||
| 41 | + | ||
| 42 | +def aclnn_index_fill_tensor_golden(self, dim, index, value, out=None, **kwargs): | ||
| 43 | + if hasattr(dim, "item"): | ||
| 44 | + dim = dim.item() | ||
| 45 | + if hasattr(value, "item"): | ||
| 46 | + value = value.item() | ||
| 47 | + if not isinstance(index, torch.Tensor): | ||
| 48 | + index = torch.tensor(index) | ||
| 49 | + result = self.clone() | ||
| 50 | + result.index_fill_(dim, index.long(), value) | ||
| 51 | + return [result] | ||
| 52 | + | ||
| 53 | + | ||
| 54 | +def aclnn_inplace_index_fill_tensor_golden(selfRef, dim, index, value, **kwargs): | ||
| 55 | + if hasattr(dim, "item"): | ||
| 56 | + dim = dim.item() | ||
| 57 | + if hasattr(value, "item"): | ||
| 58 | + value = value.item() | ||
| 59 | + if not isinstance(index, torch.Tensor): | ||
| 60 | + index = torch.tensor(index) | ||
| 61 | + result = selfRef.clone() | ||
| 62 | + result.index_fill_(dim, index.long(), value) | ||
| 63 | + return [result] | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,attributes,tensor_view_shapes,scalar_dtypes,scalar_data_ranges,output_tensor_indexes,input_data_ranges,precision_tolerances,strict_precision_mode,is_enabled | ||
| 2 | +aclnnIndexFillTensor_fp16_N_000001,aclnnIndexFillTensor,"('float32','float32')","('ND',)","{'dim':-2,'index':[1, 2]}","((7,52),(7,52))","('float32', )","((79,79),)","(1,)","((-2,-1))","((0.001, 0.001),)",,1 | ||
| 3 | +aclnnIndexFillTensor_fp16_N_000002,aclnnIndexFillTensor,"('float16','float16')","('ND',)","{'dim':0,'index':[0]}","((4,8,16),(4,8,16))","('float16', )","((10,10),)","(1,)","((0,1))","((0.001, 0.001),)",,1 | ||
| 4 | +aclnnIndexFillTensor_int32_N_000003,aclnnIndexFillTensor,"('int32','int32')","('ND',)","{'dim':1,'index':[0,1,2]}","((5,10,20),(5,10,20))","('int32', )","((100,100),)","(1,)","((-100,100))","((0.001, 0.001),)",,1 | ||
| 5 | +aclnnIndexFillTensor_int64_N_000004,aclnnIndexFillTensor,"('int64','int64')","('ND',)","{'dim':-1,'index':[3]}","((8,16,32),(8,16,32))","('int64', )","((50,50),)","(1,)","((-50,50))","((0.001, 0.001),)",,1 | ||
| 6 | +aclnnIndexFillTensor_fp16_N_000005,aclnnIndexFillTensor,"('float16','float16')","('ND',)","{'dim':0,'index':[0,2,4]}","((6,12,24),(6,12,24))","('float16', )","((-5,-5),)","(1,)","((-1,1))","((0.001, 0.001),)",,1 | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,attributes,tensor_view_shapes,output_inplace_indexes,scalar_dtypes,scalar_data_ranges,input_data_ranges,precision_tolerances,is_enabled | ||
| 2 | +aclnnInplaceIndexFillTensor_fp16_N_000001,aclnnInplaceIndexFillTensor,"('float32',)","('ND',)","{'dim':-2,'index':[1, 2]}","((7,52),)","(0,)","('float32', )","((79,79),)","((-2,-1))","((0.001, 0.001),)",1 | ||
| 3 | +aclnnInplaceIndexFillTensor_fp16_N_000002,aclnnInplaceIndexFillTensor,"('float16',)","('ND',)","{'dim':0,'index':[0]}","((4,8,16),)","(0,)","('float16', )","((10,10),)","((0,1))","((0.001, 0.001),)",1 | ||
| 4 | +aclnnInplaceIndexFillTensor_int32_N_000003,aclnnInplaceIndexFillTensor,"('int32',)","('ND',)","{'dim':1,'index':[0,1,2]}","((5,10,20),)","(0,)","('int32', )","((100,100),)","((-100,100))","((0.001, 0.001),)",1 | ||
| 5 | +aclnnInplaceIndexFillTensor_int64_N_000004,aclnnInplaceIndexFillTensor,"('int64',)","('ND',)","{'dim':-1,'index':[3]}","((8,16,32),)","(0,)","('int64', )","((50,50),)","((-50,50))","((0.001, 0.001),)",1 | ||
| 6 | +aclnnInplaceIndexFillTensor_fp16_N_000005,aclnnInplaceIndexFillTensor,"('float16',)","('ND',)","{'dim':0,'index':[0,2,4]}","((6,12,24),)","(0,)","('float16', )","((-5,-5),)","((-1,1))","((0.001, 0.001),)",1 | ||
| @@ -10,13 +10,18 @@ | |||
| 10 | # See LICENSE in the root of the software repository for the full text of the License. | 10 | # See LICENSE in the root of the software repository for the full text of the License. |
| 11 | # ---------------------------------------------------------------------------- | 11 | # ---------------------------------------------------------------------------- |
| 12 | 12 | ||
| 13 | -import numpy as np | 13 | +import torch |
| 14 | 14 | ||
| 15 | -__golden__ = {"kernel": {"masked_scatter": "masked_scatter_golden"}} | 15 | +__golden__ = { |
| 16 | + "aclnn": { | ||
| 17 | + "aclnnInplaceMaskedScatter": "aclnn_inplace_masked_scatter_golden", | ||
| 18 | + }, | ||
| 19 | + "kernel": {"masked_scatter": "masked_scatter_golden"}, | ||
| 20 | +} | ||
| 16 | 21 | ||
| 17 | 22 | ||
| 18 | def masked_scatter_golden(input0, input1, input2, **kwargs): | 23 | def masked_scatter_golden(input0, input1, input2, **kwargs): |
| 19 | - ''' | 24 | + """ |
| 20 | Golden function for masked_scatter. | 25 | Golden function for masked_scatter. |
| 21 | All the parameters (names and order) follow @masked_scatter_def.cpp without outputs. | 26 | All the parameters (names and order) follow @masked_scatter_def.cpp without outputs. |
| 22 | All the input Tensors are numpy.ndarray. | 27 | All the input Tensors are numpy.ndarray. |
| @@ -27,8 +32,7 @@ def masked_scatter_golden(input0, input1, input2, **kwargs): | |||
| 27 | 32 | ||
| 28 | Returns: | 33 | Returns: |
| 29 | Output tensor | 34 | Output tensor |
| 30 | - ''' | 35 | + """ |
| 31 | - import torch | ||
| 32 | 36 | ||
| 33 | dtype = input0.dtype | 37 | dtype = input0.dtype |
| 34 | if "bfloat16" in str(dtype): | 38 | if "bfloat16" in str(dtype): |
| @@ -44,3 +48,9 @@ def masked_scatter_golden(input0, input1, input2, **kwargs): | |||
| 44 | if "bfloat16" in str(dtype): | 48 | if "bfloat16" in str(dtype): |
| 45 | res = res.view(dtype) | 49 | res = res.view(dtype) |
| 46 | return res | 50 | return res |
| 51 | + | ||
| 52 | + | ||
| 53 | +def aclnn_inplace_masked_scatter_golden(selfRef, mask, source, **kwargs): | ||
| 54 | + result = selfRef.clone() | ||
| 55 | + result.masked_scatter_(mask.bool(), source) | ||
| 56 | + return [result] | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_view_shapes,tensor_dtypes,output_tensor_indexes,precision_tolerances,input_data_ranges,is_enabled | ||
| 2 | +aclnnMaskedScatter_001,aclnnInplaceMaskedScatter,"((3, 4), (3, 4), (3, 4))","('float', 'bool', 'float')","(0,)",,, | ||
| 3 | +aclnnMaskedScatter_002,aclnnInplaceMaskedScatter,"((3, 4, 5), (3, 4, 5), (3, 4, 5))","('float16', 'bool', 'float16')","(0,)",,, | ||
| 4 | +aclnnMaskedScatter_003,aclnnInplaceMaskedScatter,"((3, 4), (3, 4), (3, 4))","('double', 'bool', 'double')","(0,)",,, | ||
| 5 | +aclnnMaskedScatter_004,aclnnInplaceMaskedScatter,"((3, 4), (3, 4), (3, 4))","('uint8', 'bool', 'uint8')","(0,)",,, | ||
| 6 | +aclnnMaskedScatter_005,aclnnInplaceMaskedScatter,"((3, 4), (3, 4), (3, 4))","('int8', 'bool', 'int8')","(0,)",,, | ||
| @@ -11,12 +11,18 @@ | |||
| 11 | # ---------------------------------------------------------------------------- | 11 | # ---------------------------------------------------------------------------- |
| 12 | 12 | ||
| 13 | import numpy as np | 13 | import numpy as np |
| 14 | +import torch | ||
| 14 | 15 | ||
| 15 | -__golden__ = {"kernel": {"scatter": "scatter_golden"}} | 16 | +__golden__ = { |
| 17 | + "aclnn": { | ||
| 18 | + "aclnnInplaceScatterUpdate": "aclnn_inplace_scatter_update_golden", | ||
| 19 | + }, | ||
| 20 | + "kernel": {"scatter": "scatter_golden"}, | ||
| 21 | +} | ||
| 16 | 22 | ||
| 17 | 23 | ||
| 18 | def scatter_golden(var, indices, update_value, *, axis=0, **kwargs): | 24 | def scatter_golden(var, indices, update_value, *, axis=0, **kwargs): |
| 19 | - ''' | 25 | + """ |
| 20 | Golden function for scatter. | 26 | Golden function for scatter. |
| 21 | All the parameters (names and order) follow @scatter_def.cpp without outputs. | 27 | All the parameters (names and order) follow @scatter_def.cpp without outputs. |
| 22 | All the input Tensors are numpy.ndarray. | 28 | All the input Tensors are numpy.ndarray. |
| @@ -27,32 +33,43 @@ def scatter_golden(var, indices, update_value, *, axis=0, **kwargs): | |||
| 27 | 33 | ||
| 28 | Returns: | 34 | Returns: |
| 29 | Output tensor | 35 | Output tensor |
| 30 | - ''' | 36 | + """ |
| 31 | import copy | 37 | import copy |
| 32 | - | 38 | + |
| 33 | - dtype_dict = {"float32": 4, "int8": 1, "int32": 4, "float16": 2, "int64": 8, "bfloat16": 2} | 39 | + dtype_dict = { |
| 40 | + "float32": 4, | ||
| 41 | + "int8": 1, | ||
| 42 | + "int32": 4, | ||
| 43 | + "float16": 2, | ||
| 44 | + "int64": 8, | ||
| 45 | + "bfloat16": 2, | ||
| 46 | + } | ||
| 34 | all_shape = len(var.shape) | 47 | all_shape = len(var.shape) |
| 35 | abs_axis = axis | 48 | abs_axis = axis |
| 36 | if axis < 0: | 49 | if axis < 0: |
| 37 | abs_axis = all_shape + axis | 50 | abs_axis = all_shape + axis |
| 38 | - | 51 | + |
| 39 | - if not (all_shape == 4 and abs_axis == 3 and var.shape[2] % dtype_dict.get(str(var.dtype), 1) == 0 | 52 | + if not ( |
| 40 | - and var.shape[3] % dtype_dict.get(str(var.dtype), 1) == 0): | 53 | + all_shape == 4 |
| 54 | + and abs_axis == 3 | ||
| 55 | + and var.shape[2] % dtype_dict.get(str(var.dtype), 1) == 0 | ||
| 56 | + and var.shape[3] % dtype_dict.get(str(var.dtype), 1) == 0 | ||
| 57 | + ): | ||
| 41 | trans_shape_0 = var.shape[0] | 58 | trans_shape_0 = var.shape[0] |
| 42 | update_shape_0 = update_value.shape[0] | 59 | update_shape_0 = update_value.shape[0] |
| 43 | - | 60 | + |
| 44 | seceond_dim = 1 | 61 | seceond_dim = 1 |
| 45 | update_second_dim = 1 | 62 | update_second_dim = 1 |
| 46 | - | 63 | + |
| 47 | for i in range(1, abs_axis): | 64 | for i in range(1, abs_axis): |
| 48 | seceond_dim *= var.shape[i] | 65 | seceond_dim *= var.shape[i] |
| 49 | update_second_dim *= update_value.shape[i] | 66 | update_second_dim *= update_value.shape[i] |
| 50 | trans_shape_1 = seceond_dim | 67 | trans_shape_1 = seceond_dim |
| 51 | update_shape_1 = update_second_dim | 68 | update_shape_1 = update_second_dim |
| 52 | - | 69 | + |
| 53 | trans_shape_2 = var.shape[abs_axis] | 70 | trans_shape_2 = var.shape[abs_axis] |
| 54 | update_shape_2 = update_value.shape[abs_axis] | 71 | update_shape_2 = update_value.shape[abs_axis] |
| 55 | - | 72 | + |
| 56 | fourth_dim = 1 | 73 | fourth_dim = 1 |
| 57 | update_fourth_dim = 1 | 74 | update_fourth_dim = 1 |
| 58 | for i in range(abs_axis + 1, all_shape): | 75 | for i in range(abs_axis + 1, all_shape): |
| @@ -60,27 +77,33 @@ def scatter_golden(var, indices, update_value, *, axis=0, **kwargs): | |||
| 60 | update_fourth_dim *= update_value.shape[i] | 77 | update_fourth_dim *= update_value.shape[i] |
| 61 | trans_shape_3 = fourth_dim | 78 | trans_shape_3 = fourth_dim |
| 62 | update_shape_3 = update_fourth_dim | 79 | update_shape_3 = update_fourth_dim |
| 63 | - | 80 | + |
| 64 | var = var.reshape(trans_shape_0, trans_shape_1, trans_shape_2, trans_shape_3) | 81 | var = var.reshape(trans_shape_0, trans_shape_1, trans_shape_2, trans_shape_3) |
| 65 | - update_value = update_value.reshape(update_shape_0, update_shape_1, update_shape_2, update_shape_3) | 82 | + update_value = update_value.reshape( |
| 66 | - | 83 | + update_shape_0, update_shape_1, update_shape_2, update_shape_3 |
| 84 | + ) | ||
| 85 | + | ||
| 67 | axis = 2 | 86 | axis = 2 |
| 68 | - | 87 | + |
| 69 | shape_0 = update_value.shape[0] | 88 | shape_0 = update_value.shape[0] |
| 70 | shape_2 = update_value.shape[2] | 89 | shape_2 = update_value.shape[2] |
| 71 | shape_3 = update_value.shape[3] | 90 | shape_3 = update_value.shape[3] |
| 72 | output = copy.deepcopy(var) | 91 | output = copy.deepcopy(var) |
| 73 | indices_value = indices.astype(np.int64) | 92 | indices_value = indices.astype(np.int64) |
| 74 | - | 93 | + |
| 75 | if len(indices.shape) == 2: | 94 | if len(indices.shape) == 2: |
| 76 | if axis == -2 or axis == 2: | 95 | if axis == -2 or axis == 2: |
| 77 | for i in range(indices.shape[0]): | 96 | for i in range(indices.shape[0]): |
| 78 | for k in range(shape_2): | 97 | for k in range(shape_2): |
| 79 | - output[indices_value[i][0], :, indices_value[i][1] + k, :] = update_value[i, :, k, :] | 98 | + output[indices_value[i][0], :, indices_value[i][1] + k, :] = ( |
| 99 | + update_value[i, :, k, :] | ||
| 100 | + ) | ||
| 80 | elif axis == -1 or axis == 3: | 101 | elif axis == -1 or axis == 3: |
| 81 | for i in range(indices.shape[0]): | 102 | for i in range(indices.shape[0]): |
| 82 | - for l in range(shape_3): | 103 | + for idx_l in range(shape_3): |
| 83 | - output[indices_value[i][0], :, :, indices_value[i][1] + l] = update_value[i, :, :, l] | 104 | + output[indices_value[i][0], :, :, indices_value[i][1] + idx_l] = ( |
| 105 | + update_value[i, :, :, idx_l] | ||
| 106 | + ) | ||
| 84 | else: | 107 | else: |
| 85 | if axis == -2 or axis == 2: | 108 | if axis == -2 or axis == 2: |
| 86 | for i in range(shape_0): | 109 | for i in range(shape_0): |
| @@ -90,7 +113,28 @@ def scatter_golden(var, indices, update_value, *, axis=0, **kwargs): | |||
| 90 | elif axis == -1 or axis == 3: | 113 | elif axis == -1 or axis == 3: |
| 91 | for i in range(shape_0): | 114 | for i in range(shape_0): |
| 92 | indices_key = indices_value[i] | 115 | indices_key = indices_value[i] |
| 93 | - for l in range(shape_3): | 116 | + for idx_l in range(shape_3): |
| 94 | - output[i, :, :, indices_key + l] = update_value[i, :, :, l] | 117 | + output[i, :, :, indices_key + idx_l] = update_value[i, :, :, idx_l] |
| 95 | - | 118 | + |
| 96 | return output | 119 | return output |
| 120 | + | ||
| 121 | + | ||
| 122 | +def aclnn_inplace_scatter_update_golden(data, indices, updates, axis, **kwargs): | ||
| 123 | + if hasattr(axis, "item"): | ||
| 124 | + axis = axis.item() | ||
| 125 | + | ||
| 126 | + orig_dtype = data.dtype | ||
| 127 | + | ||
| 128 | + if orig_dtype == torch.bfloat16: | ||
| 129 | + data_np = data.to(torch.float32).numpy() | ||
| 130 | + updates_np = updates.to(torch.float32).numpy() | ||
| 131 | + else: | ||
| 132 | + data_np = data.numpy() | ||
| 133 | + updates_np = updates.numpy() | ||
| 134 | + | ||
| 135 | + indices_np = indices.to(torch.int64).numpy() | ||
| 136 | + | ||
| 137 | + result_np = scatter_golden(data_np, indices_np, updates_np, axis=axis) | ||
| 138 | + | ||
| 139 | + result = torch.from_numpy(result_np).to(orig_dtype) | ||
| 140 | + return [result] | ||
| @@ -0,0 +1,5 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,attributes,tensor_view_shapes,input_data_ranges,output_dtypes,output_tensor_indexes | ||
| 2 | +aclnnInplaceScatterUpdate_test_case_0,aclnnInplaceScatterUpdate,"('float32', 'int64', 'float32')","('ND',)",{ 'axis': -5},"((4, 3, 2, 5, 6, 5, 6, 5), (4,), (4, 3, 2, 2, 6, 5, 6, 5))","((-1, -0.01), (0, 3), (-2, -1))",('float32'),"(0,)" | ||
| 3 | +aclnnInplaceScatterUpdate_test_case_1,aclnnInplaceScatterUpdate,"('float16', 'int32', 'float16')","('ND',)",{ 'axis': -4},"((1, 1, 1, 1, 1), (1,), (1, 1, 1, 1, 1))","((-65504.0, -65504.0), (0, 0), (-1, -0.01))",('float16'),"(0,)" | ||
| 4 | +aclnnInplaceScatterUpdate_test_case_2,aclnnInplaceScatterUpdate,"('float16', 'int32', 'float16')","('ND',)",{ 'axis': -1},"((1, 1), (1, 2), (1, 1))","((-2, -1), (0, 0), (-2, -1))",('float16'),"(0,)" | ||
| 5 | +aclnnInplaceScatterUpdate_test_case_3,aclnnInplaceScatterUpdate,"('bfloat16', 'int64', 'bfloat16')","('ND',)",{ 'axis': 5},"((7, 7, 8, 8, 7, 9, 9, 7), (1, 2), (1, 7, 8, 8, 7, 9, 9, 7))","((-10, -2), (0, 0), (-1000, -10))",('bfloat16'),"(0,)" | ||
| @@ -11,9 +11,15 @@ | |||
| 11 | # ---------------------------------------------------------------------------- | 11 | # ---------------------------------------------------------------------------- |
| 12 | 12 | ||
| 13 | import numpy as np | 13 | import numpy as np |
| 14 | +import torch | ||
| 14 | 15 | ||
| 15 | 16 | ||
| 16 | -__golden__ = {"kernel": {"adaptive_max_pool3d": "adaptive_max_pool3d_golden"}} | 17 | +__golden__ = { |
| 18 | + "aclnn": { | ||
| 19 | + "aclnnAdaptiveMaxPool3d": "aclnn_adaptive_max_pool3d_golden", | ||
| 20 | + }, | ||
| 21 | + "kernel": {"adaptive_max_pool3d": "adaptive_max_pool3d_golden"}, | ||
| 22 | +} | ||
| 17 | 23 | ||
| 18 | 24 | ||
| 19 | def adaptive_max_pool3d_golden(x, output_size, indices_dtype=3, **kwargs): | 25 | def adaptive_max_pool3d_golden(x, output_size, indices_dtype=3, **kwargs): |
| @@ -41,3 +47,16 @@ def adaptive_max_pool3d_golden(x, output_size, indices_dtype=3, **kwargs): | |||
| 41 | indices = indices.to(torch.int32).numpy() ## 与竞品差异 | 47 | indices = indices.to(torch.int32).numpy() ## 与竞品差异 |
| 42 | 48 | ||
| 43 | return output, indices | 49 | return output, indices |
| 50 | + | ||
| 51 | + | ||
| 52 | +def aclnn_adaptive_max_pool3d_golden( | ||
| 53 | + self, outputSize=0, outputOut=None, indicesOut=None, **kwargs | ||
| 54 | +): | ||
| 55 | + if hasattr(outputSize, "tolist"): | ||
| 56 | + outputSize = outputSize.tolist() | ||
| 57 | + elif isinstance(outputSize, int): | ||
| 58 | + outputSize = [outputSize] * 3 | ||
| 59 | + result = torch.nn.functional.adaptive_max_pool3d( | ||
| 60 | + self, outputSize, return_indices=True | ||
| 61 | + ) | ||
| 62 | + return [result[0], result[1]] | ||
| @@ -0,0 +1,6 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,attributes,tensor_view_shapes,scalar_dtypes,scalar_data_ranges,output_tensor_indexes,input_data_ranges,precision_tolerances,is_enabled | ||
| 2 | +aclnn_adaptive_max_pool3d_test_00001,aclnnAdaptiveMaxPool3d,"('bfloat16', 'bfloat16', 'int32')","('ND',)","{'outputSize': [57, 8, 37]}","([0, 100, 71, 82], [0, 57, 8, 37], [0, 57, 8, 37])",(),"((None, None),)","(1, 2)","[[-100, 100]]",,TRUE | ||
| 3 | +aclnn_adaptive_max_pool3d_test_00002,aclnnAdaptiveMaxPool3d,"('bfloat16', 'bfloat16', 'int64')","('NCDHW',)","{'outputSize': [6, 13, 19]}","([0, 19, 19, 21, 23], [0, 19, 6, 13, 19], [0, 19, 6, 13, 19])",(),"((None, None),)","(1, 2)","[[-100, 100]]",,TRUE | ||
| 4 | +aclnn_adaptive_max_pool3d_test_00003,aclnnAdaptiveMaxPool3d,"('float32', 'float32', 'int32')","('ND',)","{'outputSize': [17, 3, 15]}","([47, 32, 6, 22], [47, 17, 3, 15], [47, 17, 3, 15])",(),"((None, None),)","(1, 2)","[[inf, inf]]",,TRUE | ||
| 5 | +aclnn_adaptive_max_pool3d_test_00004,aclnnAdaptiveMaxPool3d,"('float16', 'float16', 'int64')","('ND',)","{'outputSize': [2, 148, 704]}","([5, 5, 153, 8191], [5, 2, 148, 704], [5, 2, 148, 704])",(),"((None, None),)","(1, 2)","[[-inf, inf]]",,TRUE | ||
| 6 | +aclnn_adaptive_max_pool3d_test_00005,aclnnAdaptiveMaxPool3d,"('float32', 'float32', 'int32')","('ND',)","{'outputSize': [116, 198, 2]}","([2, 298, 352, 2], [2, 116, 198, 2], [2, 116, 198, 2])",(),"((None, None),)","(1, 2)","[[-inf, -inf]]",,TRUE | ||
| @@ -11,8 +11,14 @@ | |||
| 11 | # ---------------------------------------------------------------------------- | 11 | # ---------------------------------------------------------------------------- |
| 12 | 12 | ||
| 13 | import numpy as np | 13 | import numpy as np |
| 14 | +import torch | ||
| 14 | 15 | ||
| 15 | -__golden__ = {"kernel": {"ascend_quant_v2": "ascend_quant_v2_golden"}} | 16 | +__golden__ = { |
| 17 | + "aclnn": { | ||
| 18 | + "aclnnAscendQuantV3": "aclnn_ascend_quant_v3_golden", | ||
| 19 | + }, | ||
| 20 | + "kernel": {"ascend_quant_v2": "ascend_quant_v2_golden"}, | ||
| 21 | +} | ||
| 16 | 22 | ||
| 17 | 23 | ||
| 18 | def ascend_quant_v2_golden( | 24 | def ascend_quant_v2_golden( |
| @@ -83,3 +89,28 @@ def ascend_quant_v2_golden( | |||
| 83 | round_data = round_data.astype(hifloat8, copy=False) | 89 | round_data = round_data.astype(hifloat8, copy=False) |
| 84 | 90 | ||
| 85 | return round_data | 91 | return round_data |
| 92 | + | ||
| 93 | + | ||
| 94 | +def aclnn_ascend_quant_v3_golden( | ||
| 95 | + x, scale, offset, sqrtMode, roundMode, dstType, axis, y, **kwargs | ||
| 96 | +): | ||
| 97 | + if hasattr(sqrtMode, "item"): | ||
| 98 | + sqrtMode = bool(sqrtMode.item()) | ||
| 99 | + if hasattr(roundMode, "decode"): | ||
| 100 | + roundMode = roundMode.decode() | ||
| 101 | + x_f = x.to(torch.float32) | ||
| 102 | + scale_f = scale.to(torch.float32) | ||
| 103 | + offset_f = offset.to(torch.float32) if offset is not None else None | ||
| 104 | + scale_rst = x_f * (scale_f**2) if sqrtMode else x_f * scale_f | ||
| 105 | + if offset_f is not None: | ||
| 106 | + scale_rst = scale_rst + offset_f | ||
| 107 | + if roundMode == "round": | ||
| 108 | + round_data = torch.round(scale_rst) | ||
| 109 | + elif roundMode == "floor": | ||
| 110 | + round_data = torch.floor(scale_rst) | ||
| 111 | + elif roundMode == "ceil": | ||
| 112 | + round_data = torch.ceil(scale_rst) | ||
| 113 | + else: | ||
| 114 | + round_data = torch.round(scale_rst) | ||
| 115 | + round_data = round_data.clamp(-128, 127).to(torch.int8) | ||
| 116 | + return [round_data] | ||
| @@ -0,0 +1,2 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,attributes,input_data_ranges,output_tensor_indexes,absolute_precision | ||
| 2 | +aclnnAscendQuantV3_float16_ND_fuzz_1,aclnnAscendQuantV3,"('float16', 'float16', 'float16', 'int8')","('ND',)","((16, 1024),(1024,),(1024,),(16, 1024))","{'dstType': 2, 'sqrtMode':False, 'roundMode':'round', 'axis':1}","((-2, 2), (-2, 2), (-2, 2))","(-1,)",0.0001 | ||
| @@ -11,8 +11,15 @@ | |||
| 11 | # ---------------------------------------------------------------------------- | 11 | # ---------------------------------------------------------------------------- |
| 12 | 12 | ||
| 13 | import numpy as np | 13 | import numpy as np |
| 14 | +import torch | ||
| 14 | 15 | ||
| 15 | -__golden__ = {"kernel": {"dynamic_quant": "dynamic_quant_golden"}} | 16 | +__golden__ = { |
| 17 | + "aclnn": { | ||
| 18 | + "aclnnDynamicQuantV3": "aclnn_dynamic_quant_v3_golden", | ||
| 19 | + "aclnnDynamicQuant": "aclnn_dynamic_quant_golden", | ||
| 20 | + }, | ||
| 21 | + "kernel": {"dynamic_quant": "dynamic_quant_golden"}, | ||
| 22 | +} | ||
| 16 | 23 | ||
| 17 | 24 | ||
| 18 | def _dynamic_quant_common( | 25 | def _dynamic_quant_common( |
| @@ -247,3 +254,82 @@ def dynamic_quant_golden( | |||
| 247 | is_symmetrical, | 254 | is_symmetrical, |
| 248 | output_dtype_str, | 255 | output_dtype_str, |
| 249 | ) | 256 | ) |
| 257 | + | ||
| 258 | + | ||
| 259 | +def aclnn_dynamic_quant_golden(x, smoothScalesOptional, yOut, scaleOut, **kwargs): | ||
| 260 | + """ | ||
| 261 | + Aclnn golden for aclnnDynamicQuant. | ||
| 262 | + """ | ||
| 263 | + x_f = x.to(torch.float32) if x.dtype != torch.float32 else x | ||
| 264 | + smooth_scales = ( | ||
| 265 | + smoothScalesOptional.to(torch.float32) | ||
| 266 | + if smoothScalesOptional is not None | ||
| 267 | + else None | ||
| 268 | + ) | ||
| 269 | + x_scaled = x_f * smooth_scales if smooth_scales is not None else x_f | ||
| 270 | + amax = torch.amax( | ||
| 271 | + torch.abs(x_scaled).view(-1, x_scaled.shape[-1]), dim=-1, keepdim=True | ||
| 272 | + ) | ||
| 273 | + scale = amax / 127.0 | ||
| 274 | + scale = torch.where(scale == 0, torch.ones_like(scale), scale) | ||
| 275 | + quantized = torch.round(x_scaled / scale) | ||
| 276 | + quantized = quantized.clamp(-128, 127).to(torch.int8) | ||
| 277 | + return [quantized, scale] | ||
| 278 | + | ||
| 279 | + | ||
| 280 | +def aclnn_dynamic_quant_v3_golden( | ||
| 281 | + x, | ||
| 282 | + smoothScalesOptional, | ||
| 283 | + groupIndexOptional, | ||
| 284 | + dstType, | ||
| 285 | + isSymmetrical, | ||
| 286 | + quantMode, | ||
| 287 | + yOut, | ||
| 288 | + scaleOut, | ||
| 289 | + offsetOut, | ||
| 290 | + **kwargs, | ||
| 291 | +): | ||
| 292 | + """ | ||
| 293 | + Aclnn golden for aclnnDynamicQuantV3. | ||
| 294 | + """ | ||
| 295 | + if hasattr(dstType, "item"): | ||
| 296 | + dstType = dstType.item() | ||
| 297 | + if hasattr(isSymmetrical, "item"): | ||
| 298 | + isSymmetrical = bool(isSymmetrical.item()) | ||
| 299 | + if hasattr(quantMode, "item"): | ||
| 300 | + quantMode = quantMode.item() | ||
| 301 | + x_f = x.to(torch.float32) if x.dtype != torch.float32 else x | ||
| 302 | + smooth_scales = ( | ||
| 303 | + smoothScalesOptional.to(torch.float32) | ||
| 304 | + if smoothScalesOptional is not None | ||
| 305 | + else None | ||
| 306 | + ) | ||
| 307 | + x_scaled = x_f * smooth_scales if smooth_scales is not None else x_f | ||
| 308 | + scale_max = 127.0 | ||
| 309 | + scale_max_no_sym = 255.0 | ||
| 310 | + offset = None | ||
| 311 | + if not isSymmetrical: | ||
| 312 | + input_max = torch.max(x_scaled, dim=-1, keepdim=True).values | ||
| 313 | + input_min = torch.min(x_scaled, dim=-1, keepdim=True).values | ||
| 314 | + scale = (input_max - input_min) / scale_max_no_sym | ||
| 315 | + scale = torch.where(scale == 0, torch.ones_like(scale), scale) | ||
| 316 | + offset = scale_max - (input_max / scale) | ||
| 317 | + input_scaled = x_scaled / scale + offset | ||
| 318 | + else: | ||
| 319 | + input_abs = torch.abs(x_scaled) | ||
| 320 | + input_max = torch.max(input_abs, dim=-1, keepdim=True).values | ||
| 321 | + scale = input_max / scale_max | ||
| 322 | + scale = torch.where(scale == 0, torch.ones_like(scale), scale) | ||
| 323 | + input_scaled = x_scaled / scale | ||
| 324 | + round_data = torch.round(input_scaled) | ||
| 325 | + if dstType == 2: | ||
| 326 | + if isSymmetrical: | ||
| 327 | + round_data = round_data.clamp(-128, 127).to(torch.int8) | ||
| 328 | + else: | ||
| 329 | + round_data = round_data.clamp(0, 255).to(torch.uint8) | ||
| 330 | + scale_out = scale.squeeze(-1) | ||
| 331 | + if offset is not None: | ||
| 332 | + offset_out = offset.squeeze(-1) | ||
| 333 | + else: | ||
| 334 | + offset_out = torch.zeros_like(scale_out) | ||
| 335 | + return [round_data, scale_out, offset_out] | ||
| @@ -0,0 +1,2 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,absolute_precision | ||
| 2 | +aclnnDynamicQuant_float16_ND_fuzz_1,aclnnDynamicQuant,"('float16', 'float16', 'int8', 'float32')","('ND',)","((16, 1024),(1024,),(16, 1024),(16,))","((-2, 2), (-2, 2))","(2, 3)",0.0001 | ||
| @@ -11,8 +11,14 @@ | |||
| 11 | # ---------------------------------------------------------------------------- | 11 | # ---------------------------------------------------------------------------- |
| 12 | 12 | ||
| 13 | import numpy as np | 13 | import numpy as np |
| 14 | +import torch | ||
| 14 | 15 | ||
| 15 | -__golden__ = {"kernel": {"quantize": "quantize_golden"}} | 16 | +__golden__ = { |
| 17 | + "aclnn": { | ||
| 18 | + "aclnnQuantize": "aclnn_quantize_golden", | ||
| 19 | + }, | ||
| 20 | + "kernel": {"quantize": "quantize_golden"}, | ||
| 21 | +} | ||
| 16 | 22 | ||
| 17 | 23 | ||
| 18 | def quantize_golden(x, scales, zero_points=None, *, dtype, axis=1, **kwargs): | 24 | def quantize_golden(x, scales, zero_points=None, *, dtype, axis=1, **kwargs): |
| @@ -84,3 +90,25 @@ def quantize_golden(x, scales, zero_points=None, *, dtype, axis=1, **kwargs): | |||
| 84 | round_data = round_data.astype(hifloat8, copy=False) | 90 | round_data = round_data.astype(hifloat8, copy=False) |
| 85 | 91 | ||
| 86 | return round_data | 92 | return round_data |
| 93 | + | ||
| 94 | + | ||
| 95 | +def aclnn_quantize_golden(x, scales, zeroPoints, dtype, axis, out, **kwargs): | ||
| 96 | + if hasattr(dtype, "item"): | ||
| 97 | + dtype = dtype.item() | ||
| 98 | + if hasattr(axis, "item"): | ||
| 99 | + axis = axis.item() | ||
| 100 | + x_f = x.to(torch.float32) | ||
| 101 | + scale = scales.to(torch.float32) | ||
| 102 | + offset = zeroPoints.to(torch.float32) if zeroPoints is not None else None | ||
| 103 | + x_shape = x_f.shape | ||
| 104 | + scale_shape = scale.shape | ||
| 105 | + if len(x_shape) != len(scale_shape): | ||
| 106 | + tmp_scale_shape = [1] * len(x_shape) | ||
| 107 | + tmp_scale_shape[axis] = scale_shape[0] | ||
| 108 | + scale = scale.reshape(tmp_scale_shape) | ||
| 109 | + scale_rst = x_f / scale | ||
| 110 | + if offset is not None: | ||
| 111 | + offset = offset.reshape(scale.shape) | ||
| 112 | + scale_rst = scale_rst + offset | ||
| 113 | + round_data = torch.round(scale_rst).clamp(-128, 127).to(torch.int8) | ||
| 114 | + return [round_data] | ||
| @@ -0,0 +1,3 @@ | |||
| 1 | +testcase_name,api_name,tensor_dtypes,tensor_formats,tensor_view_shapes,input_data_ranges,output_tensor_indexes,attributes,absolute_precision | ||
| 2 | +aclnn_quantize_float32_ND_fuzz_1,aclnnQuantize,"('bfloat16', 'bfloat16', 'bfloat16', 'int8')","('ND',)","((19, 2, 3), (3,), (3,), (19, 2, 3))","((-10, 10),)","(-1,)","{'dtype':2, 'axis': -1}",0.0001 | ||
| 3 | +aclnn_quantize_float32_ND_fuzz_2,aclnnQuantize,"('float32', 'float16', 'int32', 'int8')","('ND',)","((19, 2, 3), (1,), (1,), (19, 2, 3))","((-10, 10),)","(-1,)","{'dtype':2, 'axis': -1}",0.0001 | ||