已合并
Fixed failed tests #26323
haiyan8创建于 2025年11月6日
Fixed failed tests #26323
已合并
共 8 个文件变更+12-11
| @@ -44,7 +44,7 @@ def exec_ut(files): | |||
| 44 | 44 | ||
| 45 | def enqueue_output(out, log_queue): | 45 | def enqueue_output(out, log_queue): |
| 46 | for line in iter(out.readline, b''): | 46 | for line in iter(out.readline, b''): |
| 47 | - log_queue.put(line.decode('utf-8', 'ignore')) | 47 | + log_queue.put(line.decode('utf-8', errors='ignore')) |
| 48 | out.close() | 48 | out.close() |
| 49 | return | 49 | return |
| 50 | 50 | ||
| @@ -78,7 +78,7 @@ class TestMode(TestCase): | |||
| 78 | def test_diff_dtype(self): | 78 | def test_diff_dtype(self): |
| 79 | path = os.path.join(os.path.dirname(__file__), '_fault_mode_cases/error_diff_dtype.py') | 79 | path = os.path.join(os.path.dirname(__file__), '_fault_mode_cases/error_diff_dtype.py') |
| 80 | process = subprocess.Popen(["torchrun", "--nproc-per-node=2", f"{path}"], shell=False, stdout=subprocess.PIPE, | 80 | process = subprocess.Popen(["torchrun", "--nproc-per-node=2", f"{path}"], shell=False, stdout=subprocess.PIPE, |
| 81 | - stderr=subprocess.PIPE, text=True) | 81 | + stderr=subprocess.PIPE, text=True, errors='ignore') |
| 82 | message = process.stderr.read() | 82 | message = process.stderr.read() |
| 83 | process.stderr.close() | 83 | process.stderr.close() |
| 84 | process.stdout.close() | 84 | process.stdout.close() |
| @@ -123,7 +123,7 @@ class TestMode(TestCase): | |||
| 123 | def test_discontinuous_tensor(self): | 123 | def test_discontinuous_tensor(self): |
| 124 | path = os.path.join(os.path.dirname(__file__), '_fault_mode_cases/error_discontinuous_tensor.py') | 124 | path = os.path.join(os.path.dirname(__file__), '_fault_mode_cases/error_discontinuous_tensor.py') |
| 125 | process = subprocess.Popen(["torchrun", "--nproc-per-node=2", f"{path}"], shell=False, stdout=subprocess.PIPE, | 125 | process = subprocess.Popen(["torchrun", "--nproc-per-node=2", f"{path}"], shell=False, stdout=subprocess.PIPE, |
| 126 | - stderr=subprocess.PIPE, text=True) | 126 | + stderr=subprocess.PIPE, text=True, errors='ignore') |
| 127 | message = process.stderr.read() | 127 | message = process.stderr.read() |
| 128 | process.stderr.close() | 128 | process.stderr.close() |
| 129 | process.stdout.close() | 129 | process.stdout.close() |
| @@ -57,7 +57,7 @@ class CombinedFlattenXCopyToContiguous(TestCase): | |||
| 57 | # case 1: flatten+strideslice ==> can be optimized as slice(contiguous with offset) + select | 57 | # case 1: flatten+strideslice ==> can be optimized as slice(contiguous with offset) + select |
| 58 | with torch.autograd.profiler.profile(use_device='npu') as prof: | 58 | with torch.autograd.profiler.profile(use_device='npu') as prof: |
| 59 | npu_out1 = npu_input.flatten()[2:100:10].contiguous() | 59 | npu_out1 = npu_input.flatten()[2:100:10].contiguous() |
| 60 | - self.assertEqual(check_operators_in_prof(['contiguous_d_Reshape', 'contiguous_d_StridedSlice'], prof) | 60 | + self.assertEqual(check_operators_in_prof(['contiguous_d_Reshape', 'contiguous_d_AsStrided'], prof) |
| 61 | or check_operators_in_prof(['aclnnInplaceCopy'], prof), | 61 | or check_operators_in_prof(['aclnnInplaceCopy'], prof), |
| 62 | True, message="Error operators called!") | 62 | True, message="Error operators called!") |
| 63 | cpu_out1 = cpu_input.flatten()[2:100:10].contiguous() | 63 | cpu_out1 = cpu_input.flatten()[2:100:10].contiguous() |
| @@ -73,7 +73,7 @@ class CombinedReshapeXCopyToContiguous(TestCase): | |||
| 73 | .view(npu_input.size(0), npu_input.size(1) * npu_input.size(2), npu_input.size(3)) \ | 73 | .view(npu_input.size(0), npu_input.size(1) * npu_input.size(2), npu_input.size(3)) \ |
| 74 | .select(2, 1) \ | 74 | .select(2, 1) \ |
| 75 | .contiguous() | 75 | .contiguous() |
| 76 | - self.assertEqual(check_operators_in_prof(['contiguous_h_match', 'contiguous_d_StridedSlice'], prof) or | 76 | + self.assertEqual(check_operators_in_prof(['contiguous_h_match', 'contiguous_d_AsStrided'], prof) or |
| 77 | check_operators_in_prof(['aclnnInplaceCopy'], prof), | 77 | check_operators_in_prof(['aclnnInplaceCopy'], prof), |
| 78 | True, message="Error operators called!") | 78 | True, message="Error operators called!") |
| 79 | cpu_out1 = cpu_input \ | 79 | cpu_out1 = cpu_input \ |
| @@ -88,7 +88,7 @@ class CombinedSqueezeXCopyToContiguous(TestCase): | |||
| 88 | # case 1: squeeze+select | 88 | # case 1: squeeze+select |
| 89 | with torch.autograd.profiler.profile(use_device='npu') as prof: | 89 | with torch.autograd.profiler.profile(use_device='npu') as prof: |
| 90 | npu_out1 = npu_input.squeeze().select(2, 1).contiguous() | 90 | npu_out1 = npu_input.squeeze().select(2, 1).contiguous() |
| 91 | - self.assertEqual(check_operators_in_prof(['contiguous_h_match', 'contiguous_d_StridedSlice'], prof) | 91 | + self.assertEqual(check_operators_in_prof(['contiguous_h_match', 'contiguous_d_AsStrided'], prof) |
| 92 | or check_operators_in_prof(['aclnnInplaceCopy'], prof), | 92 | or check_operators_in_prof(['aclnnInplaceCopy'], prof), |
| 93 | True, message="Error operators called!") | 93 | True, message="Error operators called!") |
| 94 | cpu_out1 = cpu_input.squeeze().select(2, 1).contiguous() | 94 | cpu_out1 = cpu_input.squeeze().select(2, 1).contiguous() |
| @@ -54,7 +54,7 @@ class CombinedViewsCopyToContiguous(TestCase): | |||
| 54 | # case 1: permute+select | 54 | # case 1: permute+select |
| 55 | with torch.autograd.profiler.profile(use_device='npu') as prof: | 55 | with torch.autograd.profiler.profile(use_device='npu') as prof: |
| 56 | npu_out1 = npu_input.permute(1, 3, 2, 0).select(1, 2).contiguous() | 56 | npu_out1 = npu_input.permute(1, 3, 2, 0).select(1, 2).contiguous() |
| 57 | - self.assertEqual(check_operators_in_prof(['contiguous_d_StridedSlice', 'contiguous_d_Transpose'], prof) | 57 | + self.assertEqual(check_operators_in_prof(['contiguous_d_AsStrided'], prof) |
| 58 | or check_operators_in_prof(['aclnnInplaceCopy'], prof), | 58 | or check_operators_in_prof(['aclnnInplaceCopy'], prof), |
| 59 | True, message="Error operators called!") | 59 | True, message="Error operators called!") |
| 60 | cpu_out1 = cpu_input.permute(1, 3, 2, 0).select(1, 2).contiguous() | 60 | cpu_out1 = cpu_input.permute(1, 3, 2, 0).select(1, 2).contiguous() |
| @@ -173,7 +173,7 @@ class CombinedViewsCopyToContiguous(TestCase): | |||
| 173 | # narrow at 0 dim and strideslice at last dim==> can be optimized as slice(contiguous)+select | 173 | # narrow at 0 dim and strideslice at last dim==> can be optimized as slice(contiguous)+select |
| 174 | with torch.autograd.profiler.profile(use_device='npu') as prof: | 174 | with torch.autograd.profiler.profile(use_device='npu') as prof: |
| 175 | npu_out3 = npu_input[2:4, :, :, ::2].contiguous() | 175 | npu_out3 = npu_input[2:4, :, :, ::2].contiguous() |
| 176 | - self.assertEqual(check_operators_in_prof(['contiguous_d_Reshape', 'contiguous_d_StridedSlice'], prof) | 176 | + self.assertEqual(check_operators_in_prof(['contiguous_d_Reshape', 'contiguous_d_AsStrided'], prof) |
| 177 | or check_operators_in_prof(['aclnnInplaceCopy'], prof), | 177 | or check_operators_in_prof(['aclnnInplaceCopy'], prof), |
| 178 | True, message="Error operators called!") | 178 | True, message="Error operators called!") |
| 179 | cpu_out3 = cpu_input[2:4, :, :, ::2].contiguous() | 179 | cpu_out3 = cpu_input[2:4, :, :, ::2].contiguous() |
| @@ -128,9 +128,10 @@ class SingleViewCopyToContiguous(TestCase): | |||
| 128 | for dim in range(1, len(item[2])): | 128 | for dim in range(1, len(item[2])): |
| 129 | with torch.autograd.profiler.profile(use_device='npu') as prof: | 129 | with torch.autograd.profiler.profile(use_device='npu') as prof: |
| 130 | npu_out = npu_input.select(dim, 1).contiguous() | 130 | npu_out = npu_input.select(dim, 1).contiguous() |
| 131 | - self.assertEqual(check_operators_in_prof(['contiguous_d_StridedSlice'], prof) | 131 | + self.assertEqual(check_operators_in_prof(['contiguous_d_AsStrided'], prof) |
| 132 | + or check_operators_in_prof(['contiguous_d_StridedSlice'], prof) | ||
| 132 | or check_operators_in_prof(['aclnnInplaceCopy'], prof), | 133 | or check_operators_in_prof(['aclnnInplaceCopy'], prof), |
| 133 | - True, "contiguous_d_StridedSlice or aclnnInplaceCopy is not called!") | 134 | + True, "contiguous_d_AsStrided or contiguous_d_StridedSlice or aclnnInplaceCopy is not called!") |
| 134 | cpu_out = cpu_input.select(dim, 1).contiguous() | 135 | cpu_out = cpu_input.select(dim, 1).contiguous() |
| 135 | self.assertRtolEqual(npu_out.to("cpu").numpy(), cpu_out.numpy()) | 136 | self.assertRtolEqual(npu_out.to("cpu").numpy(), cpu_out.numpy()) |
| 136 | 137 | ||
| @@ -47,7 +47,7 @@ private: | |||
| 47 | // recover src tensor info: shape and stride | 47 | // recover src tensor info: shape and stride |
| 48 | c10::SmallVector<int64_t, MAX_DIM> temp_size; | 48 | c10::SmallVector<int64_t, MAX_DIM> temp_size; |
| 49 | c10::SmallVector<int64_t, MAX_DIM> temp_stride; | 49 | c10::SmallVector<int64_t, MAX_DIM> temp_stride; |
| 50 | - for (size_t i = 0U; i <= select_size.size(); i++) { | 50 | + for (size_t i = 0U; i < select_size.size(); i++) { |
| 51 | if (base_size[i] != select_size[i] || | 51 | if (base_size[i] != select_size[i] || |
| 52 | base_stride[i] != select_stride[i]) { | 52 | base_stride[i] != select_stride[i]) { |
| 53 | temp_size.emplace_back(base_size[i]); | 53 | temp_size.emplace_back(base_size[i]); |