已合并
【2026 HCCL通信库创新大赛-粤港澳赛区】【超能陆战队】初赛及决赛代码归档 #931
liangjinxuan-2026创建于 8月3日
【2026 HCCL通信库创新大赛-粤港澳赛区】【超能陆战队】初赛及决赛代码归档 #931
已合并
共 41 个文件变更+4110-0
| @@ -0,0 +1,106 @@ | |||
| 1 | +# ----------------------------------------------------------------------------------------------------------- | ||
| 2 | +# Copyright (c) 2026 Huawei Technologies Co., Ltd. | ||
| 3 | +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of | ||
| 4 | +# CANN Open Software License Agreement Version 2.0 (the "License"). | ||
| 5 | +# Please refer to the License for details. You may not use this file except in compliance with the License. | ||
| 6 | +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, | ||
| 7 | +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. | ||
| 8 | +# See LICENSE in the root of the software repository for the full text of the License. | ||
| 9 | +# ----------------------------------------------------------------------------------------------------------- | ||
| 10 | + | ||
| 11 | +Language: Cpp | ||
| 12 | +Standard: c++17 | ||
| 13 | +TabWidth: 4 | ||
| 14 | +UseTab: Never | ||
| 15 | +UseCRLF: false | ||
| 16 | +AccessModifierOffset: -4 | ||
| 17 | +AlignConsecutiveAssignments: false | ||
| 18 | +AlignConsecutiveDeclarations: false | ||
| 19 | +AlignEscapedNewlines: DontAlign | ||
| 20 | +AlignOperands: true | ||
| 21 | +AlignTrailingComments: true | ||
| 22 | +AllowAllArgumentsOnNextLine: true | ||
| 23 | +AllowShortLambdasOnASingleLine: Empty | ||
| 24 | +AllowAllParametersOfDeclarationOnNextLine: false | ||
| 25 | +AllowShortBlocksOnASingleLine: false | ||
| 26 | +AllowShortCaseLabelsOnASingleLine: false | ||
| 27 | +AllowShortFunctionsOnASingleLine: false | ||
| 28 | +AllowShortIfStatementsOnASingleLine: false | ||
| 29 | +AllowShortLoopsOnASingleLine: false | ||
| 30 | +AlwaysBreakAfterDefinitionReturnType: None | ||
| 31 | +AlwaysBreakBeforeMultilineStrings: false | ||
| 32 | +AlwaysBreakTemplateDeclarations: MultiLine | ||
| 33 | +BinPackArguments: true | ||
| 34 | +BinPackParameters: true | ||
| 35 | +AlignAfterOpenBracket: DontAlign | ||
| 36 | + | ||
| 37 | +BraceWrapping: | ||
| 38 | + AfterCaseLabel: false | ||
| 39 | + AfterClass: false | ||
| 40 | + AfterControlStatement: Never | ||
| 41 | + AfterEnum: false | ||
| 42 | + AfterFunction: true | ||
| 43 | + AfterNamespace: false | ||
| 44 | + AfterStruct: false | ||
| 45 | + AfterUnion: false | ||
| 46 | + AfterExternBlock: false | ||
| 47 | + BeforeCatch: false | ||
| 48 | + BeforeElse: false | ||
| 49 | + IndentBraces: false | ||
| 50 | + SplitEmptyFunction: true | ||
| 51 | + SplitEmptyRecord: true | ||
| 52 | + SplitEmptyNamespace: true | ||
| 53 | + | ||
| 54 | +BreakBeforeBinaryOperators: All | ||
| 55 | +BreakBeforeBraces: Custom | ||
| 56 | +BreakBeforeTernaryOperators: true | ||
| 57 | +BreakConstructorInitializersBeforeComma: false | ||
| 58 | +BreakInheritanceList: AfterColon | ||
| 59 | +ColumnLimit: 120 | ||
| 60 | +CommentPragmas: '^ IWYU pragma:' | ||
| 61 | +PackConstructorInitializers: CurrentLine | ||
| 62 | +ConstructorInitializerIndentWidth: 4 | ||
| 63 | +ContinuationIndentWidth: 4 | ||
| 64 | +Cpp11BracedListStyle: true | ||
| 65 | +DerivePointerAlignment: false | ||
| 66 | +DisableFormat: false | ||
| 67 | +ExperimentalAutoDetectBinPacking: false | ||
| 68 | +ForEachMacros: [ foreach ] | ||
| 69 | + | ||
| 70 | +IncludeCategories: | ||
| 71 | + - Regex: '^<' | ||
| 72 | + Priority: 3 | ||
| 73 | + - Regex: '^"hccl' | ||
| 74 | + Priority: 2 | ||
| 75 | + - Regex: '.*' | ||
| 76 | + Priority: 1 | ||
| 77 | + | ||
| 78 | +IndentCaseLabels: true | ||
| 79 | +IndentWidth: 4 | ||
| 80 | +IndentWrappedFunctionNames: false | ||
| 81 | +KeepEmptyLinesAtTheStartOfBlocks: false | ||
| 82 | +MacroBlockBegin: '' | ||
| 83 | +MacroBlockEnd: '' | ||
| 84 | +MaxEmptyLinesToKeep: 1 | ||
| 85 | +NamespaceIndentation: Inner | ||
| 86 | + | ||
| 87 | +PenaltyBreakBeforeFirstCallParameter: 19 | ||
| 88 | +PenaltyBreakComment: 300 | ||
| 89 | +PenaltyBreakFirstLessLess: 120 | ||
| 90 | +PenaltyBreakString: 1000 | ||
| 91 | +PenaltyExcessCharacter: 1000000 | ||
| 92 | +PenaltyReturnTypeOnItsOwnLine: 60 | ||
| 93 | + | ||
| 94 | +PointerAlignment: Right | ||
| 95 | +ReflowComments: true | ||
| 96 | +SortIncludes: false | ||
| 97 | +SpaceAfterCStyleCast: false | ||
| 98 | +SpaceBeforeAssignmentOperators: true | ||
| 99 | +SpaceBeforeParens: ControlStatements | ||
| 100 | +SpaceInEmptyParentheses: false | ||
| 101 | +SpacesBeforeTrailingComments: 1 | ||
| 102 | +SpacesInAngles: false | ||
| 103 | +SpacesInContainerLiterals: true | ||
| 104 | +SpacesInCStyleCastParentheses: false | ||
| 105 | +SpacesInParentheses: false | ||
| 106 | +SpacesInSquareBrackets: false | ||
| @@ -0,0 +1,11 @@ | |||
| 1 | +# ----------------------------------------------------------------------------------------------------------- | ||
| 2 | +# Copyright (c) 2026 Huawei Technologies Co., Ltd. | ||
| 3 | +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of | ||
| 4 | +# CANN Open Software License Agreement Version 2.0 (the "License"). | ||
| 5 | +# Please refer to the License for details. You may not use this file except in compliance with the License. | ||
| 6 | +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, | ||
| 7 | +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. | ||
| 8 | +# See LICENSE in the root of the software repository for the full text of the License. | ||
| 9 | +# ----------------------------------------------------------------------------------------------------------- | ||
| 10 | + | ||
| 11 | +* text=auto eol=lf | ||
| @@ -0,0 +1,35 @@ | |||
| 1 | +# ----------------------------------------------------------------------------------------------------------- | ||
| 2 | +# Copyright (c) 2026 Huawei Technologies Co., Ltd. | ||
| 3 | +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of | ||
| 4 | +# CANN Open Software License Agreement Version 2.0 (the "License"). | ||
| 5 | +# Please refer to the License for details. You may not use this file except in compliance with the License. | ||
| 6 | +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, | ||
| 7 | +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. | ||
| 8 | +# See LICENSE in the root of the software repository for the full text of the License. | ||
| 9 | +# ----------------------------------------------------------------------------------------------------------- | ||
| 10 | + | ||
| 11 | +# IDE | ||
| 12 | +.vscode/ | ||
| 13 | +.idea/ | ||
| 14 | +.cache/ | ||
| 15 | +.DS_Store | ||
| 16 | +*~ | ||
| 17 | +*.swp | ||
| 18 | +*.swo | ||
| 19 | + | ||
| 20 | +# Build | ||
| 21 | +build/ | ||
| 22 | +build_ut/ | ||
| 23 | +third_party/ | ||
| 24 | +__pycache__/ | ||
| 25 | + | ||
| 26 | +# Agent | ||
| 27 | +.omo/ | ||
| 28 | +.claude/ | ||
| 29 | +.opencode/ | ||
| 30 | +CLAUDE.md | ||
| 31 | +AGENTS.md | ||
| 32 | + | ||
| 33 | +# Logs | ||
| 34 | +logs/ | ||
| 35 | +*.log | ||
| @@ -0,0 +1,24 @@ | |||
| 1 | +# ----------------------------------------------------------------------------------------------------------- | ||
| 2 | +# Copyright (c) 2026 Huawei Technologies Co., Ltd. | ||
| 3 | +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of | ||
| 4 | +# CANN Open Software License Agreement Version 2.0 (the "License"). | ||
| 5 | +# Please refer to the License for details. You may not use this file except in compliance with the License. | ||
| 6 | +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, | ||
| 7 | +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. | ||
| 8 | +# See LICENSE in the root of the software repository for the full text of the License. | ||
| 9 | +# ----------------------------------------------------------------------------------------------------------- | ||
| 10 | + | ||
| 11 | +cmake_minimum_required(VERSION 3.16.0) | ||
| 12 | +project(hccl_reduce_scatter) | ||
| 13 | + | ||
| 14 | +message(STATUS "CMAKE_BUILD_TYPE=${CMAKE_BUILD_TYPE}") | ||
| 15 | +message(STATUS "ASCEND_CANN_PACKAGE_PATH=${ASCEND_CANN_PACKAGE_PATH}") | ||
| 16 | + | ||
| 17 | +# 编译 Host 侧链接库 | ||
| 18 | +add_subdirectory(op_host) | ||
| 19 | +add_subdirectory(op_kernel_ccu) | ||
| 20 | + | ||
| 21 | +# 安装 | ||
| 22 | +install(FILES include/hccl.h | ||
| 23 | + DESTINATION include | ||
| 24 | +) | ||
| @@ -0,0 +1,201 @@ | |||
| 1 | + Apache License | ||
| 2 | + Version 2.0, January 2004 | ||
| 3 | + http://www.apache.org/licenses/ | ||
| 4 | + | ||
| 5 | + TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION | ||
| 6 | + | ||
| 7 | + 1. Definitions. | ||
| 8 | + | ||
| 9 | + "License" shall mean the terms and conditions for use, reproduction, | ||
| 10 | + and distribution as defined by Sections 1 through 9 of this document. | ||
| 11 | + | ||
| 12 | + "Licensor" shall mean the copyright owner or entity authorized by | ||
| 13 | + the copyright owner that is granting the License. | ||
| 14 | + | ||
| 15 | + "Legal Entity" shall mean the union of the acting entity and all | ||
| 16 | + other entities that control, are controlled by, or are under common | ||
| 17 | + control with that entity. For the purposes of this definition, | ||
| 18 | + "control" means (i) the power, direct or indirect, to cause the | ||
| 19 | + direction or management of such entity, whether by contract or | ||
| 20 | + otherwise, or (ii) ownership of fifty percent (50%) or more of the | ||
| 21 | + outstanding shares, or (iii) beneficial ownership of such entity. | ||
| 22 | + | ||
| 23 | + "You" (or "Your") shall mean an individual or Legal Entity | ||
| 24 | + exercising permissions granted by this License. | ||
| 25 | + | ||
| 26 | + "Source" form shall mean the preferred form for making modifications, | ||
| 27 | + including but not limited to software source code, documentation | ||
| 28 | + source, and configuration files. | ||
| 29 | + | ||
| 30 | + "Object" form shall mean any form resulting from mechanical | ||
| 31 | + transformation or translation of a Source form, including but | ||
| 32 | + not limited to compiled object code, generated documentation, | ||
| 33 | + and conversions to other media types. | ||
| 34 | + | ||
| 35 | + "Work" shall mean the work of authorship, whether in Source or | ||
| 36 | + Object form, made available under the License, as indicated by a | ||
| 37 | + copyright notice that is included in or attached to the work | ||
| 38 | + (an example is provided in the Appendix below). | ||
| 39 | + | ||
| 40 | + "Derivative Works" shall mean any work, whether in Source or Object | ||
| 41 | + form, that is based on (or derived from) the Work and for which the | ||
| 42 | + editorial revisions, annotations, elaborations, or other modifications | ||
| 43 | + represent, as a whole, an original work of authorship. For the purposes | ||
| 44 | + of this License, Derivative Works shall not include works that remain | ||
| 45 | + separable from, or merely link (or bind by name) to the interfaces of, | ||
| 46 | + the Work and Derivative Works thereof. | ||
| 47 | + | ||
| 48 | + "Contribution" shall mean any work of authorship, including | ||
| 49 | + the original version of the Work and any modifications or additions | ||
| 50 | + to that Work or Derivative Works thereof, that is intentionally | ||
| 51 | + submitted to Licensor for inclusion in the Work by the copyright owner | ||
| 52 | + or by an individual or Legal Entity authorized to submit on behalf of | ||
| 53 | + the copyright owner. For the purposes of this definition, "submitted" | ||
| 54 | + means any form of electronic, verbal, or written communication sent | ||
| 55 | + to the Licensor or its representatives, including but not limited to | ||
| 56 | + communication on electronic mailing lists, source code control systems, | ||
| 57 | + and issue tracking systems that are managed by, or on behalf of, the | ||
| 58 | + Licensor for the purpose of discussing and improving the Work, but | ||
| 59 | + excluding communication that is conspicuously marked or otherwise | ||
| 60 | + designated in writing by the copyright owner as "Not a Contribution." | ||
| 61 | + | ||
| 62 | + "Contributor" shall mean Licensor and any individual or Legal Entity | ||
| 63 | + on behalf of whom a Contribution has been received by Licensor and | ||
| 64 | + subsequently incorporated within the Work. | ||
| 65 | + | ||
| 66 | + 2. Grant of Copyright License. Subject to the terms and conditions of | ||
| 67 | + this License, each Contributor hereby grants to You a perpetual, | ||
| 68 | + worldwide, non-exclusive, no-charge, royalty-free, irrevocable | ||
| 69 | + copyright license to reproduce, prepare Derivative Works of, | ||
| 70 | + publicly display, publicly perform, sublicense, and distribute the | ||
| 71 | + Work and such Derivative Works in Source or Object form. | ||
| 72 | + | ||
| 73 | + 3. Grant of Patent License. Subject to the terms and conditions of | ||
| 74 | + this License, each Contributor hereby grants to You a perpetual, | ||
| 75 | + worldwide, non-exclusive, no-charge, royalty-free, irrevocable | ||
| 76 | + (except as stated in this section) patent license to make, have made, | ||
| 77 | + use, offer to sell, sell, import, and otherwise transfer the Work, | ||
| 78 | + where such license applies only to those patent claims licensable | ||
| 79 | + by such Contributor that are necessarily infringed by their | ||
| 80 | + Contribution(s) alone or by combination of their Contribution(s) | ||
| 81 | + with the Work to which such Contribution(s) was submitted. If You | ||
| 82 | + institute patent litigation against any entity (including a | ||
| 83 | + cross-claim or counterclaim in a lawsuit) alleging that the Work | ||
| 84 | + or a Contribution incorporated within the Work constitutes direct | ||
| 85 | + or contributory patent infringement, then any patent licenses | ||
| 86 | + granted to You under this License for that Work shall terminate | ||
| 87 | + as of the date such litigation is filed. | ||
| 88 | + | ||
| 89 | + 4. Redistribution. You may reproduce and distribute copies of the | ||
| 90 | + Work or Derivative Works thereof in any medium, with or without | ||
| 91 | + modifications, and in Source or Object form, provided that You | ||
| 92 | + meet the following conditions: | ||
| 93 | + | ||
| 94 | + (a) You must give any other recipients of the Work or | ||
| 95 | + Derivative Works a copy of this License; and | ||
| 96 | + | ||
| 97 | + (b) You must cause any modified files to carry prominent notices | ||
| 98 | + stating that You changed the files; and | ||
| 99 | + | ||
| 100 | + (c) You must retain, in the Source form of any Derivative Works | ||
| 101 | + that You distribute, all copyright, patent, trademark, and | ||
| 102 | + attribution notices from the Source form of the Work, | ||
| 103 | + excluding those notices that do not pertain to any part of | ||
| 104 | + the Derivative Works; and | ||
| 105 | + | ||
| 106 | + (d) If the Work includes a "NOTICE" text file as part of its | ||
| 107 | + distribution, then any Derivative Works that You distribute must | ||
| 108 | + include a readable copy of the attribution notices contained | ||
| 109 | + within such NOTICE file, excluding those notices that do not | ||
| 110 | + pertain to any part of the Derivative Works, in at least one | ||
| 111 | + of the following places: within a NOTICE text file distributed | ||
| 112 | + as part of the Derivative Works; within the Source form or | ||
| 113 | + documentation, if provided along with the Derivative Works; or, | ||
| 114 | + within a display generated by the Derivative Works, if and | ||
| 115 | + wherever such third-party notices normally appear. The contents | ||
| 116 | + of the NOTICE file are for informational purposes only and | ||
| 117 | + do not modify the License. You may add Your own attribution | ||
| 118 | + notices within Derivative Works that You distribute, alongside | ||
| 119 | + or as an addendum to the NOTICE text from the Work, provided | ||
| 120 | + that such additional attribution notices cannot be construed | ||
| 121 | + as modifying the License. | ||
| 122 | + | ||
| 123 | + You may add Your own copyright statement to Your modifications and | ||
| 124 | + may provide additional or different license terms and conditions | ||
| 125 | + for use, reproduction, or distribution of Your modifications, or | ||
| 126 | + for any such Derivative Works as a whole, provided Your use, | ||
| 127 | + reproduction, and distribution of the Work otherwise complies with | ||
| 128 | + the conditions stated in this License. | ||
| 129 | + | ||
| 130 | + 5. Submission of Contributions. Unless You explicitly state otherwise, | ||
| 131 | + any Contribution intentionally submitted for inclusion in the Work | ||
| 132 | + by You to the Licensor shall be under the terms and conditions of | ||
| 133 | + this License, without any additional terms or conditions. | ||
| 134 | + Notwithstanding the above, nothing herein shall supersede or modify | ||
| 135 | + the terms of any separate license agreement you may have executed | ||
| 136 | + with Licensor regarding such Contributions. | ||
| 137 | + | ||
| 138 | + 6. Trademarks. This License does not grant permission to use the trade | ||
| 139 | + names, trademarks, service marks, or product names of the Licensor, | ||
| 140 | + except as required for reasonable and customary use in describing the | ||
| 141 | + origin of the Work and reproducing the content of the NOTICE file. | ||
| 142 | + | ||
| 143 | + 7. Disclaimer of Warranty. Unless required by applicable law or | ||
| 144 | + agreed to in writing, Licensor provides the Work (and each | ||
| 145 | + Contributor provides its Contributions) on an "AS IS" BASIS, | ||
| 146 | + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or | ||
| 147 | + implied, including, without limitation, any warranties or conditions | ||
| 148 | + of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A | ||
| 149 | + PARTICULAR PURPOSE. You are solely responsible for determining the | ||
| 150 | + appropriateness of using or redistributing the Work and assume any | ||
| 151 | + risks associated with Your exercise of permissions under this License. | ||
| 152 | + | ||
| 153 | + 8. Limitation of Liability. In no event and under no legal theory, | ||
| 154 | + whether in tort (including negligence), contract, or otherwise, | ||
| 155 | + unless required by applicable law (such as deliberate and grossly | ||
| 156 | + negligent acts) or agreed to in writing, shall any Contributor be | ||
| 157 | + liable to You for damages, including any direct, indirect, special, | ||
| 158 | + incidental, or consequential damages of any character arising as a | ||
| 159 | + result of this License or out of the use or inability to use the | ||
| 160 | + Work (including but not limited to damages for loss of goodwill, | ||
| 161 | + work stoppage, computer failure or malfunction, or any and all | ||
| 162 | + other commercial damages or losses), even if such Contributor | ||
| 163 | + has been advised of the possibility of such damages. | ||
| 164 | + | ||
| 165 | + 9. Accepting Warranty or Additional Liability. While redistributing | ||
| 166 | + the Work or Derivative Works thereof, You may choose to offer, | ||
| 167 | + and charge a fee for, acceptance of support, warranty, indemnity, | ||
| 168 | + or other liability obligations and/or rights consistent with this | ||
| 169 | + License. However, in accepting such obligations, You may act only | ||
| 170 | + on Your own behalf and on Your sole responsibility, not on behalf | ||
| 171 | + of any other Contributor, and only if You agree to indemnify, | ||
| 172 | + defend, and hold each Contributor harmless for any liability | ||
| 173 | + incurred by, or claims asserted against, such Contributor by reason | ||
| 174 | + of your accepting any such warranty or additional liability. | ||
| 175 | + | ||
| 176 | + END OF TERMS AND CONDITIONS | ||
| 177 | + | ||
| 178 | + APPENDIX: How to apply the Apache License to your work. | ||
| 179 | + | ||
| 180 | + To apply the Apache License to your work, attach the following | ||
| 181 | + boilerplate notice, with the fields enclosed by brackets "{}" | ||
| 182 | + replaced with your own identifying information. (Don't include | ||
| 183 | + the brackets!) The text should be enclosed in the appropriate | ||
| 184 | + comment syntax for the file format. We also recommend that a | ||
| 185 | + file or class name and description of purpose be included on the | ||
| 186 | + same "printed page" as the copyright notice for easier | ||
| 187 | + identification within third-party archives. | ||
| 188 | + | ||
| 189 | + Copyright {yyyy} {name of copyright owner} | ||
| 190 | + | ||
| 191 | + Licensed under the Apache License, Version 2.0 (the "License"); | ||
| 192 | + you may not use this file except in compliance with the License. | ||
| 193 | + You may obtain a copy of the License at | ||
| 194 | + | ||
| 195 | + http://www.apache.org/licenses/LICENSE-2.0 | ||
| 196 | + | ||
| 197 | + Unless required by applicable law or agreed to in writing, software | ||
| 198 | + distributed under the License is distributed on an "AS IS" BASIS, | ||
| 199 | + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. | ||
| 200 | + See the License for the specific language governing permissions and | ||
| 201 | + limitations under the License. | ||
| @@ -0,0 +1,64 @@ | |||
| 1 | +# ReduceScatter 集合通信算子 | ||
| 2 | + | ||
| 3 | +## 1. 项目介绍 | ||
| 4 | + | ||
| 5 | +``` | ||
| 6 | +├── CMakeLists.txt # 顶层 CMake 配置 | ||
| 7 | +├── build.sh # 构建脚本 | ||
| 8 | +├── .clang-format # 代码风格配置 | ||
| 9 | +├── include/ # 头文件目录 | ||
| 10 | +│ ├── hccl.h # 集合通信算子头文件 | ||
| 11 | +│ ├── common.h # 通用数据结构定义 | ||
| 12 | +│ ├── custom.h # ★ 选手编写:自定义数据结构定义 | ||
| 13 | +│ ├── log.h # 日志宏定义 | ||
| 14 | +│ └── binary_stream.h # 序列化类定义 | ||
| 15 | +├── op_host/ # Host侧代码目录 | ||
| 16 | +│ ├── reduce_scatter.cc # ★ 选手编写:Host侧资源申请逻辑 | ||
| 17 | +│ └── exec_op.cc # ★ 选手编写:通信算法编排逻辑 | ||
| 18 | +└── op_kernel_ccu/ # CCU侧代码目录 | ||
| 19 | + └── ccu_kernel.cc # ★ 选手编写:通信算法编排逻辑 | ||
| 20 | +``` | ||
| 21 | + | ||
| 22 | +> [!NOTE] 注意: | ||
| 23 | +> 算子工程中已提前预制好固有逻辑,选手仅允许修改 `custom.h`、`reduce_scatter.cc`、`exec_op.h`、`exec_op.cc`、`ccu_kernel.h`、`ccu_kernel.cc` 共 6 个文件内容。 | ||
| 24 | + | ||
| 25 | +## 2. 编译运行 | ||
| 26 | + | ||
| 27 | +### 2.1 安装 CANN-Toolkit 包 | ||
| 28 | + | ||
| 29 | +请单击[下载链接](https://ascend.devcloud.huaweicloud.com/artifactory/cann-run-mirror/software/legacy/20260701000328953/),根据产品型号和环境架构下载对应软件包。安装命令如下,更多指导参考《[CANN软件安装指南](https://www.hiascend.com/document/redirect/CannCommunityInstWizard)》。 | ||
| 30 | + | ||
| 31 | +```bash | ||
| 32 | +# 确保安装包具有可执行权限 | ||
| 33 | +chmod +x Ascend-cann-toolkit_9.1.0_linux-${arch}.run | ||
| 34 | +# 安装命令 | ||
| 35 | +./Ascend-cann-toolkit_9.1.0_linux-${arch}.run --full --install-path=${install_path} | ||
| 36 | +``` | ||
| 37 | + | ||
| 38 | +### 2.2 环境变量配置 | ||
| 39 | + | ||
| 40 | +按需选择合适的命令使环境变量生效。 | ||
| 41 | + | ||
| 42 | +```bash | ||
| 43 | +# 默认路径安装,以root用户为例(非root用户,将/usr/local替换为${HOME}) | ||
| 44 | +source /usr/local/Ascend/cann/set_env.sh | ||
| 45 | +# 指定路径安装 | ||
| 46 | +# source ${install_path}/cann/set_env.sh | ||
| 47 | +``` | ||
| 48 | + | ||
| 49 | +### 2.3 编译算子工程 | ||
| 50 | + | ||
| 51 | +```bash | ||
| 52 | +bash build.sh | ||
| 53 | + | ||
| 54 | +# 编译 Debug 版本,便于断点调试 | ||
| 55 | +bash build.sh --debug | ||
| 56 | +``` | ||
| 57 | + | ||
| 58 | +## 3. 代码格式 | ||
| 59 | + | ||
| 60 | +选手代码需符合 [.clang-format](.clang-format) 文件中的代码风格规范,可通过下列命令一键修改: | ||
| 61 | + | ||
| 62 | +```bash | ||
| 63 | +bash build.sh --format | ||
| 64 | +``` | ||
| @@ -0,0 +1,10 @@ | |||
| 1 | +# 超能陆战队——决赛代码归档 | ||
| 2 | + | ||
| 3 | +- 赛区:粤港澳赛区 | ||
| 4 | +- 学校:南方科技大学 | ||
| 5 | +- 队伍:超能陆战队 | ||
| 6 | +- 成员:Sami_Hui、Qu_、Alextory-Frank | ||
| 7 | +- 赛题:ReduceScatter(CCU) | ||
| 8 | +- 归档版本:`802e278` | ||
| 9 | + | ||
| 10 | +本目录为决赛阶段最后提交的完整工程代码。实现覆盖 2×8、4×1、8+4 三种拓扑,采用双 Die 分组、分段读取流水线以及 CCU Buffer 本地归约,并保留确定性归约顺序。 | ||
| @@ -0,0 +1,101 @@ | |||
| 1 | +# ----------------------------------------------------------------------------------------------------------- | ||
| 2 | +# Copyright (c) 2026 Huawei Technologies Co., Ltd. | ||
| 3 | +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of | ||
| 4 | +# CANN Open Software License Agreement Version 2.0 (the "License"). | ||
| 5 | +# Please refer to the License for details. You may not use this file except in compliance with the License. | ||
| 6 | +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, | ||
| 7 | +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. | ||
| 8 | +# See LICENSE in the root of the software repository for the full text of the License. | ||
| 9 | +# ----------------------------------------------------------------------------------------------------------- | ||
| 10 | + | ||
| 11 | +set -e | ||
| 12 | + | ||
| 13 | +PROJECT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" | ||
| 14 | +BUILD_DIR="${PROJECT_DIR}/build" | ||
| 15 | +BUILD_TYPE="Release" | ||
| 16 | +ENABLE_FORMAT="OFF" | ||
| 17 | + | ||
| 18 | +CPU_NUM="$(nproc)" | ||
| 19 | +ASCEND_CANN_PACKAGE_PATH="/usr/local/Ascend/cann" | ||
| 20 | + | ||
| 21 | +usage() { | ||
| 22 | + cat <<EOF | ||
| 23 | +Usage: $0 [OPTIONS] | ||
| 24 | + | ||
| 25 | +Options: | ||
| 26 | + --debug 编译 Debug 版本 | ||
| 27 | + --format 格式化代码 | ||
| 28 | + -h, --help 显示帮助信息 | ||
| 29 | +EOF | ||
| 30 | + exit 0 | ||
| 31 | +} | ||
| 32 | + | ||
| 33 | +parse_args() { | ||
| 34 | + local opts | ||
| 35 | + opts=$(getopt -o h -l debug,format,help -- "$@") || usage | ||
| 36 | + eval set -- "${opts}" | ||
| 37 | + | ||
| 38 | + for arg; do | ||
| 39 | + case "${arg}" in | ||
| 40 | + --debug) BUILD_TYPE="Debug" ;; | ||
| 41 | + --format) ENABLE_FORMAT="ON" ;; | ||
| 42 | + -h|--help) usage ;; | ||
| 43 | + esac | ||
| 44 | + done | ||
| 45 | +} | ||
| 46 | + | ||
| 47 | +parse_cann_path() { | ||
| 48 | + if [[ -z "${ASCEND_HOME_PATH}" ]]; then | ||
| 49 | + printf "ERROR: ASCEND_HOME_PATH is not set.\n" >&2 | ||
| 50 | + printf "Please ensure CANN-Toolkit is properly installed and source environment variables by running:\n" >&2 | ||
| 51 | + printf " source /path/to/Ascend/cann/set_env.sh\n" >&2 | ||
| 52 | + exit 1 | ||
| 53 | + fi | ||
| 54 | + | ||
| 55 | + # 使用 local 避免污染全局(如需要全局可去掉 local) | ||
| 56 | + ASCEND_CANN_PACKAGE_PATH="${ASCEND_HOME_PATH}" | ||
| 57 | + return 0 | ||
| 58 | +} | ||
| 59 | + | ||
| 60 | +build() { | ||
| 61 | + # 创建构建目录 | ||
| 62 | + cd "${PROJECT_DIR}" | ||
| 63 | + mkdir -p "${BUILD_DIR}" | ||
| 64 | + | ||
| 65 | + # 配置 | ||
| 66 | + cmake -S . -B "${BUILD_DIR}" \ | ||
| 67 | + -DCMAKE_EXPORT_COMPILE_COMMANDS=ON \ | ||
| 68 | + -DCMAKE_BUILD_TYPE="${BUILD_TYPE}" \ | ||
| 69 | + -DCMAKE_INSTALL_PREFIX="${BUILD_DIR}" \ | ||
| 70 | + -DASCEND_CANN_PACKAGE_PATH="${ASCEND_CANN_PACKAGE_PATH}" | ||
| 71 | + | ||
| 72 | + # 编译 | ||
| 73 | + cmake --build "${BUILD_DIR}" -j${CPU_NUM} | ||
| 74 | + | ||
| 75 | + # 安装 | ||
| 76 | + cmake --install "${BUILD_DIR}" | ||
| 77 | +} | ||
| 78 | + | ||
| 79 | +format() { | ||
| 80 | + cd "${PROJECT_DIR}" | ||
| 81 | + find ./include -type f -regex '.*\.\(h\|hpp\|cpp\|cc\)$' -exec clang-format -i {} \; | ||
| 82 | + find ./op_host -type f -regex '.*\.\(h\|hpp\|cpp\|cc\)$' -exec clang-format -i {} \; | ||
| 83 | + find ./op_kernel_ccu -type f -regex '.*\.\(h\|hpp\|cpp\|cc\)$' -exec clang-format -i {} \; | ||
| 84 | +} | ||
| 85 | + | ||
| 86 | +main() { | ||
| 87 | + # 解析参数 | ||
| 88 | + parse_args "$@" | ||
| 89 | + # 解析 CANN-Toolkit 路径 | ||
| 90 | + parse_cann_path | ||
| 91 | + | ||
| 92 | + if [[ "${ENABLE_FORMAT}" == "ON" ]]; then | ||
| 93 | + # 格式化代码 | ||
| 94 | + format | ||
| 95 | + else | ||
| 96 | + # 编译 | ||
| 97 | + build | ||
| 98 | + fi | ||
| 99 | +} | ||
| 100 | + | ||
| 101 | +main "$@" | ||
A03_university/CANN-HCCL-GHM-Competition-2026/final/submissions/南方科技大学-超能陆战队/include/binary_stream.h+168-0
| @@ -0,0 +1,168 @@ | |||
| 1 | +/** | ||
| 2 | + * Copyright (c) 2026 Huawei Technologies Co., Ltd. | ||
| 3 | + * This program is free software, you can redistribute it and/or modify it under the terms and conditions of | ||
| 4 | + * CANN Open Software License Agreement Version 2.0 (the "License"). | ||
| 5 | + * Please refer to the License for details. You may not use this file except in compliance with the License. | ||
| 6 | + * THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, | ||
| 7 | + * INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. | ||
| 8 | + * See LICENSE in the root of the software repository for the full text of the License. | ||
| 9 | + */ | ||
| 10 | + | ||
| 11 | + | ||
| 12 | + | ||
| 13 | + | ||
| 14 | + | ||
| 15 | + | ||
| 16 | + | ||
| 17 | + | ||
| 18 | + | ||
| 19 | + | ||
| 20 | + | ||
| 21 | + | ||
| 22 | +/** | ||
| 23 | + * @brief 序列化工具类 | ||
| 24 | + */ | ||
| 25 | +class BinaryStream { | ||
| 26 | +public: | ||
| 27 | + static constexpr std::ios_base::openmode DEFAULT_IOS_MODE = std::ios_base::in | std::ios_base::out; | ||
| 28 | + | ||
| 29 | + explicit BinaryStream(std::ios_base::openmode mode = DEFAULT_IOS_MODE) : stream(mode | std::ios_base::binary) {}; | ||
| 30 | + | ||
| 31 | + explicit BinaryStream(std::vector<char> &buf, std::ios_base::openmode mode = DEFAULT_IOS_MODE) | ||
| 32 | + : stream(mode | std::ios_base::binary) | ||
| 33 | + { | ||
| 34 | + stream.rdbuf()->pubsetbuf(buf.data(), buf.size()); | ||
| 35 | + } | ||
| 36 | + | ||
| 37 | + template <typename T> BinaryStream &operator<<(const T &t) | ||
| 38 | + { | ||
| 39 | + stream.write(reinterpret_cast<const char *>(&t), sizeof(T)); | ||
| 40 | + return *this; | ||
| 41 | + } | ||
| 42 | + | ||
| 43 | + // 多级vector递归序列化 | ||
| 44 | + template <typename T> BinaryStream &operator<<(const std::vector<T> &vec) | ||
| 45 | + { | ||
| 46 | + size_t size = vec.size(); | ||
| 47 | + *this << size; | ||
| 48 | + for (const auto &elem : vec) { | ||
| 49 | + *this << elem; | ||
| 50 | + } | ||
| 51 | + return *this; | ||
| 52 | + } | ||
| 53 | + | ||
| 54 | + // 对string的输入函数 | ||
| 55 | + BinaryStream &operator<<(const std::string &s) | ||
| 56 | + { | ||
| 57 | + size_t size = s.size(); | ||
| 58 | + stream.write(reinterpret_cast<const char *>(&size), sizeof(size_t)); // 写入长度 | ||
| 59 | + stream.write(s.data(), size); // 写入字符数据 | ||
| 60 | + return *this; | ||
| 61 | + } | ||
| 62 | + | ||
| 63 | + template <typename T> BinaryStream &operator>>(T &t) | ||
| 64 | + { | ||
| 65 | + stream.read(reinterpret_cast<char *>(&t), sizeof(T)); | ||
| 66 | + return *this; | ||
| 67 | + } | ||
| 68 | + | ||
| 69 | + // 对string的读取函数 | ||
| 70 | + BinaryStream &operator>>(std::string &s) | ||
| 71 | + { | ||
| 72 | + size_t size; | ||
| 73 | + stream.read(reinterpret_cast<char *>(&size), sizeof(size)); // 先从流中读取字符串长度 | ||
| 74 | + s.resize(size); // 为string分配足够空间 | ||
| 75 | + stream.read(&s[0], size); // 直接读取数据到string的缓冲区中,无需再分配内存 | ||
| 76 | + return *this; | ||
| 77 | + } | ||
| 78 | + | ||
| 79 | + // 多级vector递归反序列化 | ||
| 80 | + template <typename T> BinaryStream &operator>>(std::vector<T> &vec) | ||
| 81 | + { | ||
| 82 | + size_t size; | ||
| 83 | + *this >> size; | ||
| 84 | + vec.resize(size); | ||
| 85 | + for (auto &elem : vec) { | ||
| 86 | + *this >> elem; | ||
| 87 | + } | ||
| 88 | + return *this; | ||
| 89 | + } | ||
| 90 | + | ||
| 91 | + // map序列化 | ||
| 92 | + template <typename T1, typename T2> BinaryStream &operator<<(const std::map<T1, T2> &m) | ||
| 93 | + { | ||
| 94 | + size_t size = m.size(); | ||
| 95 | + *this << size; | ||
| 96 | + for (const auto &elem : m) { | ||
| 97 | + *this << elem.first; | ||
| 98 | + *this << elem.second; | ||
| 99 | + } | ||
| 100 | + return *this; | ||
| 101 | + } | ||
| 102 | + | ||
| 103 | + // map反序列化 | ||
| 104 | + template <typename T1, typename T2> BinaryStream &operator>>(std::map<T1, T2> &m) | ||
| 105 | + { | ||
| 106 | + size_t size; | ||
| 107 | + *this >> size; | ||
| 108 | + for (size_t i = 0; i < size; i++) { | ||
| 109 | + T1 key; | ||
| 110 | + *this >> key; | ||
| 111 | + T2 value; | ||
| 112 | + *this >> value; | ||
| 113 | + m[key] = value; | ||
| 114 | + } | ||
| 115 | + return *this; | ||
| 116 | + } | ||
| 117 | + | ||
| 118 | + void Dump(std::vector<char> &vec) | ||
| 119 | + { | ||
| 120 | + std::for_each(std::istreambuf_iterator<char>(stream), std::istreambuf_iterator<char>(), [&vec](const char c) { | ||
| 121 | + vec.push_back(c); | ||
| 122 | + }); | ||
| 123 | + } | ||
| 124 | + | ||
| 125 | + void DumpWithRevert(std::vector<char> &vec) | ||
| 126 | + { | ||
| 127 | + std::streampos originalPos = stream.tellg(); // 保存原始位置 | ||
| 128 | + std::for_each(std::istreambuf_iterator<char>(stream), std::istreambuf_iterator<char>(), [&vec](const char c) { | ||
| 129 | + vec.push_back(c); | ||
| 130 | + }); | ||
| 131 | + stream.seekg(originalPos); // 恢复原始位置 | ||
| 132 | + } | ||
| 133 | + | ||
| 134 | + std::uint64_t GetSize() | ||
| 135 | + { | ||
| 136 | + return stream.str().size(); | ||
| 137 | + } | ||
| 138 | + | ||
| 139 | + std::string GetString() | ||
| 140 | + { | ||
| 141 | + return stream.str(); | ||
| 142 | + } | ||
| 143 | + | ||
| 144 | + std::string SplictStream(uint64_t &start, uint64_t &end) | ||
| 145 | + { | ||
| 146 | + std::string temp = stream.str(); | ||
| 147 | + if (start >= temp.length()) { | ||
| 148 | + HCCL_ERROR("[SplictStream]start[%lu] is bigger than stream length[%lu]", start, temp.length()); | ||
| 149 | + return ""; | ||
| 150 | + } | ||
| 151 | + | ||
| 152 | + // 截取子串 | ||
| 153 | + std::string result = temp.substr(start, end - start); | ||
| 154 | + | ||
| 155 | + // 返回新的 string | ||
| 156 | + return result; | ||
| 157 | + } | ||
| 158 | + | ||
| 159 | + void Clear() | ||
| 160 | + { | ||
| 161 | + stream.clear(); | ||
| 162 | + } | ||
| 163 | + | ||
| 164 | +private: | ||
| 165 | + std::stringstream stream; | ||
| 166 | +}; | ||
| 167 | + | ||
| 168 | + | ||
| @@ -0,0 +1,54 @@ | |||
| 1 | +/** | ||
| 2 | + * Copyright (c) 2026 Huawei Technologies Co., Ltd. | ||
| 3 | + * This program is free software, you can redistribute it and/or modify it under the terms and conditions of | ||
| 4 | + * CANN Open Software License Agreement Version 2.0 (the "License"). | ||
| 5 | + * Please refer to the License for details. You may not use this file except in compliance with the License. | ||
| 6 | + * THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, | ||
| 7 | + * INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. | ||
| 8 | + * See LICENSE in the root of the software repository for the full text of the License. | ||
| 9 | + */ | ||
| 10 | + | ||
| 11 | + | ||
| 12 | + | ||
| 13 | + | ||
| 14 | + | ||
| 15 | + | ||
| 16 | + | ||
| 17 | + | ||
| 18 | + | ||
| 19 | + | ||
| 20 | + | ||
| 21 | + | ||
| 22 | + | ||
| 23 | + | ||
| 24 | + | ||
| 25 | +constexpr uint32_t NOTIFY_IDX_ACK = 0; | ||
| 26 | +constexpr uint32_t NOTIFY_IDX_DATA_SIGNAL = 1; | ||
| 27 | +constexpr uint32_t CUSTOM_TIMEOUT = 1800; | ||
| 28 | + | ||
| 29 | +constexpr uint32_t COMM_INDENTIFIER_MAX_LENGTH = 128; | ||
| 30 | +constexpr uint32_t OP_NAME_LENGTH = 32; | ||
| 31 | +constexpr uint32_t TAG_LENGTH = OP_NAME_LENGTH + COMM_INDENTIFIER_MAX_LENGTH; | ||
| 32 | +constexpr uint32_t INVALID_VALUE_RANKID = 0xFFFFFFFF; | ||
| 33 | +constexpr uint32_t MAX_DATA_SIZE = 256 * 1024 * 1024; // 单次通信的最大数据量,256MB | ||
| 34 | +constexpr uint64_t MAX_RANK_SIZE = 16; | ||
| 35 | + | ||
| 36 | +struct OpParam { | ||
| 37 | + char tag[TAG_LENGTH]; | ||
| 38 | + void *inputPtr = nullptr; | ||
| 39 | + void *outputPtr = nullptr; | ||
| 40 | + uint64_t count = 0; | ||
| 41 | + uint32_t root = 0; | ||
| 42 | + uint32_t myRank = INVALID_VALUE_RANKID; | ||
| 43 | + uint32_t rankSize = 0; | ||
| 44 | + HcclDataType dataType = HCCL_DATA_TYPE_RESERVED; | ||
| 45 | + HcclCMDType opType = HcclCMDType::HCCL_CMD_INVALID; | ||
| 46 | + HcclReduceOp reduceType = HcclReduceOp::HCCL_REDUCE_SUM; | ||
| 47 | + ThreadHandle cpuThread; | ||
| 48 | + void *resCtx = nullptr; ///< 通信引擎上下文中的资源信息,存放 AlgResourceCtx 序列化后的内容 | ||
| 49 | + uint64_t ctxSize = 0; | ||
| 50 | +}; | ||
| 51 | + | ||
| 52 | +const std::unordered_map<HcclDataType, uint32_t> SIZE_TABLE = {{HCCL_DATA_TYPE_FP32, sizeof(float)}}; | ||
| 53 | + | ||
| 54 | + | ||
| @@ -0,0 +1,94 @@ | |||
| 1 | +/** | ||
| 2 | + * Copyright (c) 2026 Huawei Technologies Co., Ltd. | ||
| 3 | + * This program is free software, you can redistribute it and/or modify it under the terms and conditions of | ||
| 4 | + * CANN Open Software License Agreement Version 2.0 (the "License"). | ||
| 5 | + * Please refer to the License for details. You may not use this file except in compliance with the License. | ||
| 6 | + * THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, | ||
| 7 | + * INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. | ||
| 8 | + * See LICENSE in the root of the software repository for the full text of the License. | ||
| 9 | + */ | ||
| 10 | + | ||
| 11 | + | ||
| 12 | + | ||
| 13 | + | ||
| 14 | + | ||
| 15 | + | ||
| 16 | + | ||
| 17 | + | ||
| 18 | + | ||
| 19 | + | ||
| 20 | + | ||
| 21 | + | ||
| 22 | +typedef struct { | ||
| 23 | + void *addr; | ||
| 24 | + uint64_t size; | ||
| 25 | +} CommBuffer; | ||
| 26 | + | ||
| 27 | +struct CcuKernelArgBase { | ||
| 28 | + ChannelHandle channels[MAX_RANK_SIZE]; | ||
| 29 | + uint32_t channelCount; | ||
| 30 | +}; | ||
| 31 | + | ||
| 32 | +struct CcuReduceScatterKernelArg : public CcuKernelArgBase { | ||
| 33 | + uint32_t remoteRanks[MAX_RANK_SIZE]; | ||
| 34 | + uint32_t rankSize; | ||
| 35 | + uint32_t rankId; | ||
| 36 | + uint32_t scratchSlotBase; | ||
| 37 | + bool includeLocal; | ||
| 38 | + HcclDataType dataType; | ||
| 39 | + HcclReduceOp reduceOp; | ||
| 40 | +}; | ||
| 41 | + | ||
| 42 | +struct CcuCombineKernelArg : public CcuKernelArgBase {}; | ||
| 43 | + | ||
| 44 | +// ccu kernel register所需信息 | ||
| 45 | +struct CcuKernelInfo { | ||
| 46 | + // kernel名称 | ||
| 47 | + char kernelFuncName[64]; | ||
| 48 | + // kernel函数 | ||
| 49 | + void *kernelFunc; | ||
| 50 | + // KernelArg实例指针 | ||
| 51 | + void *kernelArg; | ||
| 52 | + | ||
| 53 | +private: | ||
| 54 | + std::shared_ptr<CcuKernelArgBase> kernelArgSmartPtr; | ||
| 55 | + | ||
| 56 | +public: | ||
| 57 | + template <typename T> void setKernelArg(std::shared_ptr<T> arg) | ||
| 58 | + { | ||
| 59 | + kernelArgSmartPtr = std::static_pointer_cast<CcuKernelArgBase>(arg); | ||
| 60 | + kernelArg = static_cast<void *>(arg.get()); | ||
| 61 | + } | ||
| 62 | +}; | ||
| 63 | + | ||
| 64 | +struct AlgResourceCtx { | ||
| 65 | + ThreadHandle ccuThread; ///< CCU通信引擎上的thread资源 | ||
| 66 | + CommBuffer localBuffer; ///< 本端HCCL通信内存 | ||
| 67 | + std::vector<ThreadHandle> threads; ///< CCU通信引擎上的thread资源 | ||
| 68 | + std::vector<CcuKernelHandle> ccuKernels; | ||
| 69 | + | ||
| 70 | + // 序列化 | ||
| 71 | + std::vector<char> Serialize() | ||
| 72 | + { | ||
| 73 | + BinaryStream binaryStream; | ||
| 74 | + binaryStream << ccuThread; | ||
| 75 | + binaryStream << localBuffer; | ||
| 76 | + binaryStream << threads; | ||
| 77 | + binaryStream << ccuKernels; | ||
| 78 | + std::vector<char> result; | ||
| 79 | + binaryStream.Dump(result); | ||
| 80 | + return result; | ||
| 81 | + } | ||
| 82 | + | ||
| 83 | + // 反序列化 | ||
| 84 | + void DeSerialize(std::vector<char> &data) | ||
| 85 | + { | ||
| 86 | + BinaryStream binaryStream(data); | ||
| 87 | + binaryStream >> ccuThread; | ||
| 88 | + binaryStream >> localBuffer; | ||
| 89 | + binaryStream >> threads; | ||
| 90 | + binaryStream >> ccuKernels; | ||
| 91 | + } | ||
| 92 | +}; | ||
| 93 | + | ||
| 94 | + | ||
| @@ -0,0 +1,31 @@ | |||
| 1 | +/** | ||
| 2 | + * Copyright (c) 2026 Huawei Technologies Co., Ltd. | ||
| 3 | + * This program is free software, you can redistribute it and/or modify it under the terms and conditions of | ||
| 4 | + * CANN Open Software License Agreement Version 2.0 (the "License"). | ||
| 5 | + * Please refer to the License for details. You may not use this file except in compliance with the License. | ||
| 6 | + * THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, | ||
| 7 | + * INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. | ||
| 8 | + * See LICENSE in the root of the software repository for the full text of the License. | ||
| 9 | + */ | ||
| 10 | + | ||
| 11 | + | ||
| 12 | + | ||
| 13 | + | ||
| 14 | + | ||
| 15 | + | ||
| 16 | + | ||
| 17 | + | ||
| 18 | + | ||
| 19 | + | ||
| 20 | +extern "C" { | ||
| 21 | + | ||
| 22 | + | ||
| 23 | +/* ReduceScatter 算子入口 */ | ||
| 24 | +HcclResult HcclReduceScatter(void *sendBuf, void *recvBuf, uint64_t recvCount, HcclDataType dataType, | ||
| 25 | + HcclReduceOp op, HcclComm comm, aclrtStream stream); | ||
| 26 | + | ||
| 27 | + | ||
| 28 | +} | ||
| 29 | + | ||
| 30 | + | ||
| 31 | + | ||
| @@ -0,0 +1,140 @@ | |||
| 1 | +/** | ||
| 2 | + * Copyright (c) 2026 Huawei Technologies Co., Ltd. | ||
| 3 | + * This program is free software, you can redistribute it and/or modify it under the terms and conditions of | ||
| 4 | + * CANN Open Software License Agreement Version 2.0 (the "License"). | ||
| 5 | + * Please refer to the License for details. You may not use this file except in compliance with the License. | ||
| 6 | + * THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, | ||
| 7 | + * INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. | ||
| 8 | + * See LICENSE in the root of the software repository for the full text of the License. | ||
| 9 | + */ | ||
| 10 | + | ||
| 11 | + | ||
| 12 | + | ||
| 13 | + | ||
| 14 | + | ||
| 15 | + | ||
| 16 | + | ||
| 17 | + | ||
| 18 | + | ||
| 19 | + | ||
| 20 | + | ||
| 21 | + | ||
| 22 | +typedef enum { | ||
| 23 | + LOG_LEVEL_DEBUG = 0, // DEBUG级别 | ||
| 24 | + LOG_LEVEL_INFO = 1, // INFO级别 | ||
| 25 | + LOG_LEVEL_WARNING = 2, // WARNING级别 | ||
| 26 | + LOG_LEVEL_ERROR = 3, // ERROR级别 | ||
| 27 | + LOG_LEVEL_NONE = 4 // 关闭所有日志 | ||
| 28 | +} LogLevel; | ||
| 29 | + | ||
| 30 | +// CcuResult 返回码转换为 HcclResult 返回码 | ||
| 31 | +inline HcclResult ConvertCcuToHccl(CcuResult ccuResult) | ||
| 32 | +{ | ||
| 33 | + switch (ccuResult) { | ||
| 34 | + case CCU_SUCCESS: | ||
| 35 | + return HCCL_SUCCESS; | ||
| 36 | + case CCU_E_PARA: | ||
| 37 | + return HCCL_E_PARA; | ||
| 38 | + case CCU_E_PTR: | ||
| 39 | + return HCCL_E_PTR; | ||
| 40 | + case CCU_E_INTERNAL: | ||
| 41 | + return HCCL_E_INTERNAL; | ||
| 42 | + case CCU_E_NOT_SUPPORT: | ||
| 43 | + return HCCL_E_NOT_SUPPORT; | ||
| 44 | + case CCU_E_NOT_FOUND: | ||
| 45 | + return HCCL_E_NOT_FOUND; | ||
| 46 | + case CCU_E_UNAVAIL: | ||
| 47 | + return HCCL_E_UNAVAIL; | ||
| 48 | + default: | ||
| 49 | + return HCCL_E_INTERNAL; | ||
| 50 | + } | ||
| 51 | +} | ||
| 52 | + | ||
| 53 | + | ||
| 54 | + | ||
| 55 | + | ||
| 56 | + | ||
| 57 | + | ||
| 58 | + | ||
| 59 | + do { \ | ||
| 60 | + if (LOG_LEVEL <= LOG_LEVEL_DEBUG) { \ | ||
| 61 | + printf("[DEBUG][%s][%s:%d]" format "\n", __func__, __FILE__, __LINE__, ##__VA_ARGS__); \ | ||
| 62 | + } \ | ||
| 63 | + } while (0) | ||
| 64 | + | ||
| 65 | + | ||
| 66 | + do { \ | ||
| 67 | + if (LOG_LEVEL <= LOG_LEVEL_INFO) { \ | ||
| 68 | + printf("[INFO][%s][%s:%d]" format "\n", __func__, __FILE__, __LINE__, ##__VA_ARGS__); \ | ||
| 69 | + } \ | ||
| 70 | + } while (0) | ||
| 71 | + | ||
| 72 | + | ||
| 73 | + do { \ | ||
| 74 | + if (LOG_LEVEL <= LOG_LEVEL_WARNING) { \ | ||
| 75 | + printf("[WARN][%s][%s:%d]" format "\n", __func__, __FILE__, __LINE__, ##__VA_ARGS__); \ | ||
| 76 | + } \ | ||
| 77 | + } while (0) | ||
| 78 | + | ||
| 79 | + | ||
| 80 | + do { \ | ||
| 81 | + if (LOG_LEVEL <= LOG_LEVEL_ERROR) { \ | ||
| 82 | + printf("[ERROR][%s][%s:%d]" format "\n", __func__, __FILE__, __LINE__, ##__VA_ARGS__); \ | ||
| 83 | + } \ | ||
| 84 | + } while (0) | ||
| 85 | + | ||
| 86 | +/* 检查指针, 若指针为NULL, 则记录日志, 并返回错误 */ | ||
| 87 | + | ||
| 88 | + do { \ | ||
| 89 | + if (UNLIKELY((ptr) == nullptr)) { \ | ||
| 90 | + HCCL_ERROR("[%s] ptr [%s] is nullptr, return HCCL_E_PTR", __func__, | ||
| 91 | + return HCCL_E_PTR; \ | ||
| 92 | + } \ | ||
| 93 | + } while (0) | ||
| 94 | + | ||
| 95 | +/* 检查函数返回值, 记录指定日志, 并返回指定错误码 */ | ||
| 96 | + | ||
| 97 | + do { \ | ||
| 98 | + if (UNLIKELY(result)) { \ | ||
| 99 | + exeLog; \ | ||
| 100 | + return retCode; \ | ||
| 101 | + } \ | ||
| 102 | + } while (0) | ||
| 103 | + | ||
| 104 | +/* 检查函数返回值, 并返回指定错误码 */ | ||
| 105 | + | ||
| 106 | + do { \ | ||
| 107 | + int32_t hcclRet = call; \ | ||
| 108 | + if (UNLIKELY(hcclRet != HCCL_SUCCESS)) { \ | ||
| 109 | + if (hcclRet == HCCL_E_AGAIN) { \ | ||
| 110 | + HCCL_WARNING("[%s] call trace: hcclRet -> %d", __func__, hcclRet); \ | ||
| 111 | + } else { \ | ||
| 112 | + HCCL_ERROR("[%s] call trace: hcclRet -> %d", __func__, hcclRet); \ | ||
| 113 | + } \ | ||
| 114 | + return static_cast<HcclResult>(hcclRet); \ | ||
| 115 | + } \ | ||
| 116 | + } while (0) | ||
| 117 | + | ||
| 118 | +/* 检查函数返回值, 并返回指定错误码 */ | ||
| 119 | + | ||
| 120 | + do { \ | ||
| 121 | + CcuResult ccuRet = call; \ | ||
| 122 | + if (UNLIKELY(ccuRet != CCU_SUCCESS)) { \ | ||
| 123 | + HCCL_ERROR("[%s] call trace: ccuRet -> %d", __func__, ccuRet); \ | ||
| 124 | + return ConvertCcuToHccl(ccuRet); \ | ||
| 125 | + } \ | ||
| 126 | + } while (0) | ||
| 127 | + | ||
| 128 | + | ||
| 129 | + do { \ | ||
| 130 | + aclError ret = cmd; \ | ||
| 131 | + if (UNLIKELY(ret != ACL_SUCCESS)) { \ | ||
| 132 | + HCCL_ERROR("acl interface return err %s:%d, retcode: %d.\n", __FILE__, __LINE__, ret); \ | ||
| 133 | + if (ret == ACL_ERROR_RT_MEMORY_ALLOCATION) { \ | ||
| 134 | + HCCL_ERROR("memory allocation error, check whether the current memory space is sufficient.\n"); \ | ||
| 135 | + } \ | ||
| 136 | + return HCCL_E_RUNTIME; \ | ||
| 137 | + } \ | ||
| 138 | + } while (0) | ||
| 139 | + | ||
| 140 | + | ||
A03_university/CANN-HCCL-GHM-Competition-2026/final/submissions/南方科技大学-超能陆战队/op_host/CMakeLists.txt+69-0
| @@ -0,0 +1,69 @@ | |||
| 1 | +# ----------------------------------------------------------------------------------------------------------- | ||
| 2 | +# Copyright (c) 2026 Huawei Technologies Co., Ltd. | ||
| 3 | +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of | ||
| 4 | +# CANN Open Software License Agreement Version 2.0 (the "License"). | ||
| 5 | +# Please refer to the License for details. You may not use this file except in compliance with the License. | ||
| 6 | +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, | ||
| 7 | +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. | ||
| 8 | +# See LICENSE in the root of the software repository for the full text of the License. | ||
| 9 | +# ----------------------------------------------------------------------------------------------------------- | ||
| 10 | + | ||
| 11 | +# ================================================== | ||
| 12 | +# Host 侧动态链接库 | ||
| 13 | +# ================================================== | ||
| 14 | +add_library(hccl SHARED) | ||
| 15 | + | ||
| 16 | +target_sources(hccl PRIVATE | ||
| 17 | + ${CMAKE_CURRENT_SOURCE_DIR}/reduce_scatter.cc | ||
| 18 | + ${CMAKE_CURRENT_SOURCE_DIR}/exec_op.cc | ||
| 19 | +) | ||
| 20 | + | ||
| 21 | +target_include_directories(hccl PRIVATE | ||
| 22 | + ${CMAKE_CURRENT_SOURCE_DIR}/ | ||
| 23 | + ${CMAKE_CURRENT_SOURCE_DIR}/../include | ||
| 24 | + # CANN Toolkit头文件 | ||
| 25 | + ${ASCEND_CANN_PACKAGE_PATH}/include | ||
| 26 | + ${ASCEND_CANN_PACKAGE_PATH}/include/hccl | ||
| 27 | + ${ASCEND_CANN_PACKAGE_PATH}/include/hcomm | ||
| 28 | + ${ASCEND_CANN_PACKAGE_PATH}/include/hcomm/ccu | ||
| 29 | + ${ASCEND_CANN_PACKAGE_PATH}/pkg_inc | ||
| 30 | + ${ASCEND_CANN_PACKAGE_PATH}/pkg_inc/hccl | ||
| 31 | + ${ASCEND_CANN_PACKAGE_PATH}/pkg_inc/hcomm | ||
| 32 | + ${ASCEND_CANN_PACKAGE_PATH}/pkg_inc/hcomm/ccu | ||
| 33 | +) | ||
| 34 | + | ||
| 35 | +target_compile_options(hccl PRIVATE | ||
| 36 | + -std=c++17 | ||
| 37 | + -Werror | ||
| 38 | + -Wall | ||
| 39 | + -fno-common | ||
| 40 | + -fno-strict-aliasing | ||
| 41 | + -pipe | ||
| 42 | + -fstack-protector-all | ||
| 43 | + -U_FORTIFY_SOURCE | ||
| 44 | + $<$<CONFIG:Debug>:-g -O0> | ||
| 45 | + $<$<CONFIG:Release>:-O2> | ||
| 46 | +) | ||
| 47 | + | ||
| 48 | +target_compile_definitions(hccl PRIVATE | ||
| 49 | + _GLIBCXX_USE_CXX11_ABI=0 | ||
| 50 | +) | ||
| 51 | + | ||
| 52 | +target_link_options(hccl PRIVATE | ||
| 53 | + -Wl,-z,relro | ||
| 54 | + -Wl,-z,now | ||
| 55 | + -Wl,-z,noexecstack | ||
| 56 | +) | ||
| 57 | + | ||
| 58 | +target_link_directories(hccl PRIVATE | ||
| 59 | + ${ASCEND_CANN_PACKAGE_PATH}/lib64 | ||
| 60 | +) | ||
| 61 | + | ||
| 62 | +target_link_libraries(hccl PRIVATE | ||
| 63 | + hcomm | ||
| 64 | + acl_rt | ||
| 65 | +) | ||
| 66 | + | ||
| 67 | +install(TARGETS hccl LIBRARY | ||
| 68 | + DESTINATION lib64 | ||
| 69 | +) | ||
A03_university/CANN-HCCL-GHM-Competition-2026/final/submissions/南方科技大学-超能陆战队/op_host/exec_op.cc+282-0
| @@ -0,0 +1,282 @@ | |||
| 1 | +/** | ||
| 2 | + * Copyright (c) 2026 Huawei Technologies Co., Ltd. | ||
| 3 | + * This program is free software, you can redistribute it and/or modify it under the terms and conditions of | ||
| 4 | + * CANN Open Software License Agreement Version 2.0 (the "License"). | ||
| 5 | + * Please refer to the License for details. You may not use this file except in compliance with the License. | ||
| 6 | + * THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, | ||
| 7 | + * INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. | ||
| 8 | + * See LICENSE in the root of the software repository for the full text of the License. | ||
| 9 | + */ | ||
| 10 | + | ||
| 11 | + | ||
| 12 | + | ||
| 13 | + | ||
| 14 | + | ||
| 15 | + | ||
| 16 | + | ||
| 17 | + | ||
| 18 | + | ||
| 19 | + | ||
| 20 | + | ||
| 21 | +namespace ops_hccl { | ||
| 22 | +namespace { | ||
| 23 | + constexpr uint64_t MAX_CHUNK_BYTES = 256ULL * 1024ULL * 1024ULL; | ||
| 24 | + constexpr uint64_t PARALLEL_COMBINE_MIN_BYTES = 256ULL * 1024ULL; | ||
| 25 | + constexpr uint64_t READ_PIPELINE_MIN_BYTES = 512ULL * 1024ULL; | ||
| 26 | + constexpr uint64_t READ_PIPELINE_ALIGNMENT = 4ULL * 1024ULL; | ||
| 27 | + constexpr uint64_t BUFFERED_REDUCE_MIN_BYTES = 256ULL * 1024ULL; | ||
| 28 | + constexpr uint64_t BUFFERED_REDUCE_SLICE_BYTES = 4ULL * 1024ULL; | ||
| 29 | + constexpr uint64_t BUFFERED_REDUCE_LOOP_COUNT = 16ULL; | ||
| 30 | + | ||
| 31 | + struct BufferedReduceArgs { | ||
| 32 | + uint64_t enabled = 0; | ||
| 33 | + uint64_t addressOffset = 0; | ||
| 34 | + uint64_t loopIterations = 0; | ||
| 35 | + uint64_t parallelConfig = 0; | ||
| 36 | + uint64_t residualBytes = 0; | ||
| 37 | + }; | ||
| 38 | + | ||
| 39 | + constexpr uint64_t BitMask(uint16_t bitCount) | ||
| 40 | + { | ||
| 41 | + return (uint64_t{1} << bitCount) - 1; | ||
| 42 | + } | ||
| 43 | + | ||
| 44 | + uint64_t PackParallelConfig(uint64_t repeatCount, uint64_t repeatLoopIndex, uint64_t totalLoopCount) | ||
| 45 | + { | ||
| 46 | + return ((repeatCount & BitMask(7)) << 55) | ((repeatLoopIndex & BitMask(7)) << 48) | ||
| 47 | + | ((totalLoopCount & BitMask(7)) << 41); | ||
| 48 | + } | ||
| 49 | + | ||
| 50 | + BufferedReduceArgs CalcBufferedReduceArgs(uint64_t bytes) | ||
| 51 | + { | ||
| 52 | + BufferedReduceArgs args; | ||
| 53 | + if (bytes < BUFFERED_REDUCE_MIN_BYTES) { | ||
| 54 | + return args; | ||
| 55 | + } | ||
| 56 | + | ||
| 57 | + constexpr uint64_t loopBytes = BUFFERED_REDUCE_SLICE_BYTES * BUFFERED_REDUCE_LOOP_COUNT; | ||
| 58 | + const uint64_t serialIterations = bytes / loopBytes; | ||
| 59 | + const uint64_t parallelSlices = (bytes % loopBytes) / BUFFERED_REDUCE_SLICE_BYTES; | ||
| 60 | + const uint64_t residualBytes = bytes % BUFFERED_REDUCE_SLICE_BYTES; | ||
| 61 | + | ||
| 62 | + args.enabled = 1; | ||
| 63 | + args.addressOffset = serialIterations * loopBytes; | ||
| 64 | + args.loopIterations = serialIterations; | ||
| 65 | + if (parallelSlices == 0 && residualBytes == 0) { | ||
| 66 | + return args; | ||
| 67 | + } | ||
| 68 | + if (parallelSlices != 0 && residualBytes == 0) { | ||
| 69 | + args.parallelConfig = PackParallelConfig(parallelSlices - 1, 0, 1); | ||
| 70 | + args.residualBytes = BUFFERED_REDUCE_SLICE_BYTES; | ||
| 71 | + } else if (parallelSlices == 0) { | ||
| 72 | + args.parallelConfig = PackParallelConfig(0, 0, 1); | ||
| 73 | + args.residualBytes = residualBytes; | ||
| 74 | + } else { | ||
| 75 | + args.parallelConfig = PackParallelConfig(parallelSlices - 1, 1, 2); | ||
| 76 | + args.residualBytes = residualBytes; | ||
| 77 | + } | ||
| 78 | + return args; | ||
| 79 | + } | ||
| 80 | + | ||
| 81 | + HcclResult ConvertCcuResult(CcuResult result) | ||
| 82 | + { | ||
| 83 | + switch (result) { | ||
| 84 | + case CCU_SUCCESS: | ||
| 85 | + return HCCL_SUCCESS; | ||
| 86 | + case CCU_E_PARA: | ||
| 87 | + return HCCL_E_PARA; | ||
| 88 | + case CCU_E_PTR: | ||
| 89 | + return HCCL_E_PTR; | ||
| 90 | + case CCU_E_NOT_SUPPORT: | ||
| 91 | + return HCCL_E_NOT_SUPPORT; | ||
| 92 | + case CCU_E_NOT_FOUND: | ||
| 93 | + return HCCL_E_NOT_FOUND; | ||
| 94 | + case CCU_E_UNAVAIL: | ||
| 95 | + return HCCL_E_UNAVAIL; | ||
| 96 | + default: | ||
| 97 | + return HCCL_E_INTERNAL; | ||
| 98 | + } | ||
| 99 | + } | ||
| 100 | + | ||
| 101 | + HcclResult PreSyncThreads(ThreadHandle mainThread, ThreadHandle subThread) | ||
| 102 | + { | ||
| 103 | + HcclResult result | ||
| 104 | + = static_cast<HcclResult>(HcommThreadNotifyRecordOnThread(mainThread, subThread, 0)); | ||
| 105 | + if (result != HCCL_SUCCESS) { | ||
| 106 | + return result; | ||
| 107 | + } | ||
| 108 | + return static_cast<HcclResult>(HcommThreadNotifyWaitOnThreadWithDefaultTimeout(subThread, 0)); | ||
| 109 | + } | ||
| 110 | + | ||
| 111 | + HcclResult PostSyncThreads(ThreadHandle mainThread, ThreadHandle subThread) | ||
| 112 | + { | ||
| 113 | + HcclResult result | ||
| 114 | + = static_cast<HcclResult>(HcommThreadNotifyWaitOnThreadWithDefaultTimeout(mainThread, 0)); | ||
| 115 | + if (result != HCCL_SUCCESS) { | ||
| 116 | + return result; | ||
| 117 | + } | ||
| 118 | + return static_cast<HcclResult>(HcommThreadNotifyRecordOnThread(subThread, mainThread, 0)); | ||
| 119 | + } | ||
| 120 | + | ||
| 121 | + HcclResult LaunchKernel(ThreadHandle thread, CcuKernelHandle kernel, uint64_t inputAddress, | ||
| 122 | + uint64_t destinationAddress, uint64_t inputToken, uint64_t destinationToken, uint64_t scratchAddress, | ||
| 123 | + uint64_t scratchToken, uint64_t sourceBaseOffset, uint64_t chunkBytes, uint64_t pipelineFirstBytes, | ||
| 124 | + uint64_t pipelineSecondBytes, const BufferedReduceArgs &firstReduce, | ||
| 125 | + const BufferedReduceArgs &secondReduce) | ||
| 126 | + { | ||
| 127 | + const uint64_t taskArgs[] = { | ||
| 128 | + inputAddress, destinationAddress, inputToken, destinationToken, scratchAddress, scratchToken, | ||
| 129 | + sourceBaseOffset, chunkBytes, pipelineFirstBytes, pipelineSecondBytes, | ||
| 130 | + firstReduce.enabled, firstReduce.addressOffset, firstReduce.loopIterations, | ||
| 131 | + firstReduce.parallelConfig, firstReduce.residualBytes, | ||
| 132 | + secondReduce.enabled, secondReduce.addressOffset, secondReduce.loopIterations, | ||
| 133 | + secondReduce.parallelConfig, secondReduce.residualBytes}; | ||
| 134 | + CcuResult launchResult = HcommCcuKernelLaunch(thread, kernel, taskArgs, sizeof(taskArgs) / sizeof(taskArgs[0])); | ||
| 135 | + if (launchResult != CCU_SUCCESS) { | ||
| 136 | + return ConvertCcuResult(launchResult); | ||
| 137 | + } | ||
| 138 | + return HCCL_SUCCESS; | ||
| 139 | + } | ||
| 140 | + | ||
| 141 | + HcclResult LaunchCombineKernel(ThreadHandle thread, CcuKernelHandle kernel, uint64_t destinationAddress, | ||
| 142 | + uint64_t sourceAddress, uint64_t destinationToken, uint64_t sourceToken, uint64_t chunkBytes) | ||
| 143 | + { | ||
| 144 | + const uint64_t taskArgs[] = { | ||
| 145 | + destinationAddress, sourceAddress, destinationToken, sourceToken, chunkBytes}; | ||
| 146 | + CcuResult launchResult = HcommCcuKernelLaunch(thread, kernel, taskArgs, sizeof(taskArgs) / sizeof(taskArgs[0])); | ||
| 147 | + if (launchResult != CCU_SUCCESS) { | ||
| 148 | + return ConvertCcuResult(launchResult); | ||
| 149 | + } | ||
| 150 | + return HCCL_SUCCESS; | ||
| 151 | + } | ||
| 152 | +} // namespace | ||
| 153 | + | ||
| 154 | +HcclResult ExecOp(const OpParam ¶m) | ||
| 155 | +{ | ||
| 156 | + // 反序列化 | ||
| 157 | + char *ctx = static_cast<char *>(param.resCtx); | ||
| 158 | + std::vector<char> seq(ctx, ctx + param.ctxSize); | ||
| 159 | + AlgResourceCtx resCtx; | ||
| 160 | + resCtx.DeSerialize(seq); | ||
| 161 | + | ||
| 162 | + if (param.count == 0) { | ||
| 163 | + return HCCL_SUCCESS; | ||
| 164 | + } | ||
| 165 | + if (param.rankSize == 0 || param.rankSize > MAX_RANK_SIZE) { | ||
| 166 | + return HCCL_E_PARA; | ||
| 167 | + } | ||
| 168 | + | ||
| 169 | + constexpr uint64_t dataTypeBytes = sizeof(float); | ||
| 170 | + if (param.count > std::numeric_limits<uint64_t>::max() / dataTypeBytes) { | ||
| 171 | + return HCCL_E_PARA; | ||
| 172 | + } | ||
| 173 | + const uint64_t recvBytes = param.count * dataTypeBytes; | ||
| 174 | + if (recvBytes > std::numeric_limits<uint64_t>::max() / param.rankSize) { | ||
| 175 | + return HCCL_E_PARA; | ||
| 176 | + } | ||
| 177 | + const uint64_t inputBytes = recvBytes * param.rankSize; | ||
| 178 | + | ||
| 179 | + if (param.rankSize == 1) { | ||
| 180 | + return static_cast<HcclResult>( | ||
| 181 | + HcommLocalCopyOnThread(resCtx.threads[0], param.outputPtr, param.inputPtr, recvBytes)); | ||
| 182 | + } | ||
| 183 | + if (resCtx.threads.empty() || resCtx.ccuKernels.empty()) { | ||
| 184 | + return HCCL_E_INTERNAL; | ||
| 185 | + } | ||
| 186 | + const bool useTwoDies = resCtx.threads.size() == 2 && resCtx.ccuKernels.size() == 4; | ||
| 187 | + if (!useTwoDies && (resCtx.threads.size() != 1 || resCtx.ccuKernels.size() != 1)) { | ||
| 188 | + return HCCL_E_INTERNAL; | ||
| 189 | + } | ||
| 190 | + | ||
| 191 | + const uint64_t baseInputAddress = reinterpret_cast<uint64_t>(param.inputPtr); | ||
| 192 | + const uint64_t baseOutputAddress = reinterpret_cast<uint64_t>(param.outputPtr); | ||
| 193 | + uint64_t inputToken = 0; | ||
| 194 | + uint64_t outputToken = 0; | ||
| 195 | + CcuResult tokenResult = HcommCcuGetMemToken(baseInputAddress, inputBytes, &inputToken); | ||
| 196 | + if (tokenResult != CCU_SUCCESS) { | ||
| 197 | + return ConvertCcuResult(tokenResult); | ||
| 198 | + } | ||
| 199 | + tokenResult = HcommCcuGetMemToken(baseOutputAddress, recvBytes, &outputToken); | ||
| 200 | + if (tokenResult != CCU_SUCCESS) { | ||
| 201 | + return ConvertCcuResult(tokenResult); | ||
| 202 | + } | ||
| 203 | + | ||
| 204 | + if (resCtx.localBuffer.addr == nullptr | ||
| 205 | + || resCtx.localBuffer.size < static_cast<uint64_t>(param.rankSize) * dataTypeBytes) { | ||
| 206 | + return HCCL_E_MEMORY; | ||
| 207 | + } | ||
| 208 | + | ||
| 209 | + const uint64_t scratchSlotCount = param.rankSize; | ||
| 210 | + uint64_t maxChunkBytes = std::min(MAX_CHUNK_BYTES, resCtx.localBuffer.size / scratchSlotCount); | ||
| 211 | + maxChunkBytes -= maxChunkBytes % dataTypeBytes; | ||
| 212 | + if (maxChunkBytes == 0) { | ||
| 213 | + return HCCL_E_MEMORY; | ||
| 214 | + } | ||
| 215 | + const uint64_t scratchBytes = maxChunkBytes * scratchSlotCount; | ||
| 216 | + const uint64_t scratchAddress = reinterpret_cast<uint64_t>(resCtx.localBuffer.addr); | ||
| 217 | + uint64_t scratchToken = 0; | ||
| 218 | + tokenResult = HcommCcuGetMemToken(scratchAddress, scratchBytes, &scratchToken); | ||
| 219 | + if (tokenResult != CCU_SUCCESS) { | ||
| 220 | + return ConvertCcuResult(tokenResult); | ||
| 221 | + } | ||
| 222 | + | ||
| 223 | + uint64_t processedBytes = 0; | ||
| 224 | + while (processedBytes < recvBytes) { | ||
| 225 | + const uint64_t chunkBytes = std::min(maxChunkBytes, recvBytes - processedBytes); | ||
| 226 | + const uint64_t outputAddress = baseOutputAddress + processedBytes; | ||
| 227 | + const uint64_t sourceBaseOffset = static_cast<uint64_t>(param.myRank) * recvBytes + processedBytes; | ||
| 228 | + uint64_t pipelineFirstBytes = 0; | ||
| 229 | + uint64_t pipelineSecondBytes = 0; | ||
| 230 | + if (chunkBytes >= READ_PIPELINE_MIN_BYTES) { | ||
| 231 | + pipelineFirstBytes = (chunkBytes / 2 / READ_PIPELINE_ALIGNMENT) * READ_PIPELINE_ALIGNMENT; | ||
| 232 | + if (pipelineFirstBytes == 0 || pipelineFirstBytes >= chunkBytes) { | ||
| 233 | + pipelineFirstBytes = 0; | ||
| 234 | + } else { | ||
| 235 | + pipelineSecondBytes = chunkBytes - pipelineFirstBytes; | ||
| 236 | + } | ||
| 237 | + } | ||
| 238 | + const uint64_t firstReduceBytes = pipelineFirstBytes == 0 ? chunkBytes : pipelineFirstBytes; | ||
| 239 | + const BufferedReduceArgs firstReduce = CalcBufferedReduceArgs(firstReduceBytes); | ||
| 240 | + const BufferedReduceArgs secondReduce = CalcBufferedReduceArgs(pipelineSecondBytes); | ||
| 241 | + | ||
| 242 | + if (!useTwoDies) { | ||
| 243 | + CHK_RET(LaunchKernel(resCtx.threads[0], resCtx.ccuKernels[0], baseInputAddress, outputAddress, inputToken, | ||
| 244 | + outputToken, scratchAddress, scratchToken, sourceBaseOffset, chunkBytes, pipelineFirstBytes, | ||
| 245 | + pipelineSecondBytes, firstReduce, secondReduce)); | ||
| 246 | + } else { | ||
| 247 | + const uint64_t partialAddress = scratchAddress; | ||
| 248 | + CHK_RET(PreSyncThreads(resCtx.threads[0], resCtx.threads[1])); | ||
| 249 | + CHK_RET(LaunchKernel(resCtx.threads[0], resCtx.ccuKernels[0], baseInputAddress, outputAddress, inputToken, | ||
| 250 | + outputToken, scratchAddress, scratchToken, sourceBaseOffset, chunkBytes, pipelineFirstBytes, | ||
| 251 | + pipelineSecondBytes, firstReduce, secondReduce)); | ||
| 252 | + CHK_RET(LaunchKernel(resCtx.threads[1], resCtx.ccuKernels[1], baseInputAddress, partialAddress, inputToken, | ||
| 253 | + scratchToken, scratchAddress, scratchToken, sourceBaseOffset, chunkBytes, pipelineFirstBytes, | ||
| 254 | + pipelineSecondBytes, firstReduce, secondReduce)); | ||
| 255 | + CHK_RET(PostSyncThreads(resCtx.threads[0], resCtx.threads[1])); | ||
| 256 | + | ||
| 257 | + const uint64_t firstHalfBytes = pipelineFirstBytes != 0 | ||
| 258 | + ? pipelineFirstBytes | ||
| 259 | + : (chunkBytes / dataTypeBytes / 2) * dataTypeBytes; | ||
| 260 | + const uint64_t secondHalfBytes = chunkBytes - firstHalfBytes; | ||
| 261 | + if (chunkBytes < PARALLEL_COMBINE_MIN_BYTES || firstHalfBytes == 0) { | ||
| 262 | + CHK_RET(LaunchCombineKernel(resCtx.threads[0], resCtx.ccuKernels[2], outputAddress, partialAddress, | ||
| 263 | + outputToken, scratchToken, chunkBytes)); | ||
| 264 | + } else { | ||
| 265 | + const uint64_t secondPartialAddress = pipelineFirstBytes == 0 | ||
| 266 | + ? partialAddress + firstHalfBytes | ||
| 267 | + : scratchAddress + static_cast<uint64_t>(param.rankSize) * pipelineFirstBytes; | ||
| 268 | + CHK_RET(PreSyncThreads(resCtx.threads[0], resCtx.threads[1])); | ||
| 269 | + CHK_RET(LaunchCombineKernel(resCtx.threads[0], resCtx.ccuKernels[2], outputAddress, partialAddress, | ||
| 270 | + outputToken, scratchToken, firstHalfBytes)); | ||
| 271 | + CHK_RET(LaunchCombineKernel(resCtx.threads[1], resCtx.ccuKernels[3], | ||
| 272 | + outputAddress + firstHalfBytes, secondPartialAddress, | ||
| 273 | + outputToken, scratchToken, secondHalfBytes)); | ||
| 274 | + CHK_RET(PostSyncThreads(resCtx.threads[0], resCtx.threads[1])); | ||
| 275 | + } | ||
| 276 | + } | ||
| 277 | + processedBytes += chunkBytes; | ||
| 278 | + } | ||
| 279 | + | ||
| 280 | + return HCCL_SUCCESS; | ||
| 281 | +} | ||
| 282 | +} // namespace ops_hccl | ||
| @@ -0,0 +1,21 @@ | |||
| 1 | +/** | ||
| 2 | + * Copyright (c) 2026 Huawei Technologies Co., Ltd. | ||
| 3 | + * This program is free software, you can redistribute it and/or modify it under the terms and conditions of | ||
| 4 | + * CANN Open Software License Agreement Version 2.0 (the "License"). | ||
| 5 | + * Please refer to the License for details. You may not use this file except in compliance with the License. | ||
| 6 | + * THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, | ||
| 7 | + * INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. | ||
| 8 | + * See LICENSE in the root of the software repository for the full text of the License. | ||
| 9 | + */ | ||
| 10 | + | ||
| 11 | + | ||
| 12 | + | ||
| 13 | + | ||
| 14 | + | ||
| 15 | + | ||
| 16 | + | ||
| 17 | +namespace ops_hccl { | ||
| 18 | +// 执行算法任务编排 | ||
| 19 | +HcclResult ExecOp(const OpParam ¶m); | ||
| 20 | +} // namespace ops_hccl | ||
| 21 | + | ||
| @@ -0,0 +1,343 @@ | |||
| 1 | +/** | ||
| 2 | + * Copyright (c) 2026 Huawei Technologies Co., Ltd. | ||
| 3 | + * This program is free software, you can redistribute it and/or modify it under the terms and conditions of | ||
| 4 | + * CANN Open Software License Agreement Version 2.0 (the "License"). | ||
| 5 | + * Please refer to the License for details. You may not use this file except in compliance with the License. | ||
| 6 | + * THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, | ||
| 7 | + * INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. | ||
| 8 | + * See LICENSE in the root of the software repository for the full text of the License. | ||
| 9 | + */ | ||
| 10 | + | ||
| 11 | + | ||
| 12 | + | ||
| 13 | + | ||
| 14 | + | ||
| 15 | + | ||
| 16 | + | ||
| 17 | + | ||
| 18 | + | ||
| 19 | + | ||
| 20 | + | ||
| 21 | + | ||
| 22 | + | ||
| 23 | + | ||
| 24 | + | ||
| 25 | + | ||
| 26 | + | ||
| 27 | + | ||
| 28 | + | ||
| 29 | + | ||
| 30 | +namespace { | ||
| 31 | +constexpr uint32_t CHANNEL_NOTIFY_NUM = 3; | ||
| 32 | +constexpr uint32_t MAX_DIE_NUM = 2; | ||
| 33 | + | ||
| 34 | +struct ChannelGroup { | ||
| 35 | + uint32_t dieId; | ||
| 36 | + std::vector<ChannelHandle> channels; | ||
| 37 | + std::vector<uint32_t> remoteRanks; | ||
| 38 | +}; | ||
| 39 | + | ||
| 40 | +// 用以持久化持有所有注册算子的参数指针,防止函数退出时被析构 | ||
| 41 | +static std::vector<std::shared_ptr<void>> g_persistentKernelArgs; | ||
| 42 | + | ||
| 43 | +HcclResult ConvertCcuResult(CcuResult result) | ||
| 44 | +{ | ||
| 45 | + switch (result) { | ||
| 46 | + case CCU_SUCCESS: | ||
| 47 | + return HCCL_SUCCESS; | ||
| 48 | + case CCU_E_PARA: | ||
| 49 | + return HCCL_E_PARA; | ||
| 50 | + case CCU_E_PTR: | ||
| 51 | + return HCCL_E_PTR; | ||
| 52 | + case CCU_E_NOT_SUPPORT: | ||
| 53 | + return HCCL_E_NOT_SUPPORT; | ||
| 54 | + case CCU_E_NOT_FOUND: | ||
| 55 | + return HCCL_E_NOT_FOUND; | ||
| 56 | + case CCU_E_UNAVAIL: | ||
| 57 | + return HCCL_E_UNAVAIL; | ||
| 58 | + default: | ||
| 59 | + return HCCL_E_INTERNAL; | ||
| 60 | + } | ||
| 61 | +} | ||
| 62 | + | ||
| 63 | +HcclResult AcquirePeerChannels( | ||
| 64 | + HcclComm comm, CommEngine engine, uint32_t myRank, uint32_t rankSize, std::vector<ChannelGroup> &groups) | ||
| 65 | +{ | ||
| 66 | + uint32_t *networkLayers = nullptr; | ||
| 67 | + uint32_t networkLayerCount = 0; | ||
| 68 | + CHK_RET(HcclRankGraphGetLayers(comm, &networkLayers, &networkLayerCount)); | ||
| 69 | + if (networkLayers == nullptr || networkLayerCount == 0) { | ||
| 70 | + HCCL_ERROR("[ReduceScatter] No topology layer is available for rank %u", myRank); | ||
| 71 | + return HCCL_E_NOT_FOUND; | ||
| 72 | + } | ||
| 73 | + const std::vector<uint32_t> networkLayerIds(networkLayers, networkLayers + networkLayerCount); | ||
| 74 | + | ||
| 75 | + std::vector<HcclChannelDesc> channelDescs; | ||
| 76 | + std::vector<uint32_t> remoteRanks; | ||
| 77 | + std::vector<uint32_t> channelDieIds; | ||
| 78 | + channelDescs.reserve(rankSize - 1); | ||
| 79 | + remoteRanks.reserve(rankSize - 1); | ||
| 80 | + channelDieIds.reserve(rankSize - 1); | ||
| 81 | + | ||
| 82 | + for (uint32_t remoteRank = 0; remoteRank < rankSize; ++remoteRank) { | ||
| 83 | + if (remoteRank == myRank) { | ||
| 84 | + continue; | ||
| 85 | + } | ||
| 86 | + | ||
| 87 | + HcclChannelDesc desc; | ||
| 88 | + CHK_RET(HcclChannelDescInit(&desc, 1)); | ||
| 89 | + bool protocolFound = false; | ||
| 90 | + | ||
| 91 | + for (uint32_t layerIndex = 0; layerIndex < networkLayerCount && !protocolFound; ++layerIndex) { | ||
| 92 | + uint32_t linkCount = 0; | ||
| 93 | + CommLink *links = nullptr; | ||
| 94 | + HcclResult linkResult | ||
| 95 | + = HcclRankGraphGetLinks(comm, networkLayerIds[layerIndex], myRank, remoteRank, &links, &linkCount); | ||
| 96 | + if (linkResult != HCCL_SUCCESS || links == nullptr) { | ||
| 97 | + continue; | ||
| 98 | + } | ||
| 99 | + | ||
| 100 | + for (uint32_t linkIndex = 0; linkIndex < linkCount; ++linkIndex) { | ||
| 101 | + const CommLink &link = links[linkIndex]; | ||
| 102 | + if (link.linkAttr.linkProtocol != CommProtocol::COMM_PROTOCOL_UBC_CTP) { | ||
| 103 | + continue; | ||
| 104 | + } | ||
| 105 | + | ||
| 106 | + desc.remoteRank = remoteRank; | ||
| 107 | + desc.notifyNum = CHANNEL_NOTIFY_NUM; | ||
| 108 | + desc.channelProtocol = link.linkAttr.linkProtocol; | ||
| 109 | + desc.localEndpoint.protocol = link.srcEndpointDesc.protocol; | ||
| 110 | + desc.localEndpoint.commAddr = link.srcEndpointDesc.commAddr; | ||
| 111 | + desc.localEndpoint.loc = link.srcEndpointDesc.loc; | ||
| 112 | + desc.remoteEndpoint.protocol = link.dstEndpointDesc.protocol; | ||
| 113 | + desc.remoteEndpoint.commAddr = link.dstEndpointDesc.commAddr; | ||
| 114 | + desc.remoteEndpoint.loc = link.dstEndpointDesc.loc; | ||
| 115 | + protocolFound = true; | ||
| 116 | + break; | ||
| 117 | + } | ||
| 118 | + } | ||
| 119 | + | ||
| 120 | + if (!protocolFound) { | ||
| 121 | + HCCL_ERROR("[ReduceScatter] UBC_CTP link not found between rank %u and rank %u", myRank, remoteRank); | ||
| 122 | + return HCCL_E_NOT_FOUND; | ||
| 123 | + } | ||
| 124 | + | ||
| 125 | + EndpointAttrDieId dieId = 0; | ||
| 126 | + uint32_t infoLength = sizeof(dieId); | ||
| 127 | + CHK_RET(HcclRankGraphGetEndpointInfo( | ||
| 128 | + comm, myRank, &desc.localEndpoint, ENDPOINT_ATTR_DIE_ID, infoLength, &dieId)); | ||
| 129 | + if (dieId >= MAX_DIE_NUM) { | ||
| 130 | + HCCL_ERROR("[ReduceScatter] Invalid die id %u for channel from rank %u to rank %u", dieId, myRank, | ||
| 131 | + remoteRank); | ||
| 132 | + return HCCL_E_INTERNAL; | ||
| 133 | + } | ||
| 134 | + | ||
| 135 | + channelDescs.push_back(desc); | ||
| 136 | + remoteRanks.push_back(remoteRank); | ||
| 137 | + channelDieIds.push_back(dieId); | ||
| 138 | + } | ||
| 139 | + | ||
| 140 | + std::vector<ChannelHandle> channels(channelDescs.size()); | ||
| 141 | + CHK_RET(HcclChannelAcquire( | ||
| 142 | + comm, engine, channelDescs.data(), static_cast<uint32_t>(channelDescs.size()), channels.data())); | ||
| 143 | + | ||
| 144 | + std::array<ChannelGroup, MAX_DIE_NUM> groupsByDie; | ||
| 145 | + for (uint32_t dieId = 0; dieId < MAX_DIE_NUM; ++dieId) { | ||
| 146 | + groupsByDie[dieId].dieId = dieId; | ||
| 147 | + } | ||
| 148 | + for (uint32_t channelIndex = 0; channelIndex < channels.size(); ++channelIndex) { | ||
| 149 | + ChannelGroup &group = groupsByDie[channelDieIds[channelIndex]]; | ||
| 150 | + group.channels.push_back(channels[channelIndex]); | ||
| 151 | + group.remoteRanks.push_back(remoteRanks[channelIndex]); | ||
| 152 | + } | ||
| 153 | + | ||
| 154 | + groups.clear(); | ||
| 155 | + for (uint32_t dieId = 0; dieId < MAX_DIE_NUM; ++dieId) { | ||
| 156 | + if (groupsByDie[dieId].channels.empty()) { | ||
| 157 | + continue; | ||
| 158 | + } | ||
| 159 | + groups.push_back(std::move(groupsByDie[dieId])); | ||
| 160 | + } | ||
| 161 | + if (groups.size() == MAX_DIE_NUM && groups[0].channels.size() > groups[1].channels.size()) { | ||
| 162 | + // The primary group also reduces the local rank. Put it on the die | ||
| 163 | + // with fewer peer channels to balance the number of sources. | ||
| 164 | + std::swap(groups[0], groups[1]); | ||
| 165 | + } | ||
| 166 | + | ||
| 167 | + return HCCL_SUCCESS; | ||
| 168 | +} | ||
| 169 | + | ||
| 170 | +HcclResult RegisterCcuKernels( | ||
| 171 | + HcclComm comm, const OpParam ¶m, const std::vector<ChannelGroup> &groups, AlgResourceCtx &resource) | ||
| 172 | +{ | ||
| 173 | + CcuInsHandle instructionHandle{0}; | ||
| 174 | + uint32_t instructionCount = 0; | ||
| 175 | + CHK_RET(HcclCommQueryCcuIns(comm, &instructionHandle, &instructionCount)); | ||
| 176 | + if (instructionCount != 1) { | ||
| 177 | + HCCL_ERROR("[ReduceScatter] Expected one CCU instruction instance, got %u", instructionCount); | ||
| 178 | + return HCCL_E_INTERNAL; | ||
| 179 | + } | ||
| 180 | + | ||
| 181 | + CcuResult ccuResult = HcommCcuKernelRegisterStart(instructionHandle); | ||
| 182 | + if (ccuResult != CCU_SUCCESS) { | ||
| 183 | + return ConvertCcuResult(ccuResult); | ||
| 184 | + } | ||
| 185 | + | ||
| 186 | + std::vector<std::shared_ptr<CcuReduceScatterKernelArg>> kernelArguments; | ||
| 187 | + kernelArguments.reserve(groups.size()); | ||
| 188 | + const bool needsCombineKernel = groups.size() == MAX_DIE_NUM; | ||
| 189 | + resource.ccuKernels.resize(groups.size() + (needsCombineKernel ? groups.size() : 0)); | ||
| 190 | + | ||
| 191 | + for (uint32_t groupIndex = 0; groupIndex < groups.size(); ++groupIndex) { | ||
| 192 | + const ChannelGroup &group = groups[groupIndex]; | ||
| 193 | + auto kernelArg = std::make_shared<CcuReduceScatterKernelArg>(); | ||
| 194 | + kernelArg->rankSize = param.rankSize; | ||
| 195 | + kernelArg->rankId = param.myRank; | ||
| 196 | + kernelArg->includeLocal = groupIndex == 0; | ||
| 197 | + kernelArg->scratchSlotBase | ||
| 198 | + = needsCombineKernel && groupIndex == 0 ? static_cast<uint32_t>(groups[1].channels.size()) : 0; | ||
| 199 | + kernelArg->dataType = param.dataType; | ||
| 200 | + kernelArg->reduceOp = param.reduceType; | ||
| 201 | + kernelArg->channelCount = static_cast<uint32_t>(group.channels.size()); | ||
| 202 | + for (uint32_t channelIndex = 0; channelIndex < group.channels.size(); ++channelIndex) { | ||
| 203 | + kernelArg->channels[channelIndex] = group.channels[channelIndex]; | ||
| 204 | + kernelArg->remoteRanks[channelIndex] = group.remoteRanks[channelIndex]; | ||
| 205 | + } | ||
| 206 | + | ||
| 207 | + const void *kernelArgs[] = {kernelArg.get()}; | ||
| 208 | + constexpr uint32_t registerDieId = 0; // Reserved by the registration API; channels select the actual die. | ||
| 209 | + ccuResult = HcommCcuKernelRegister(instructionHandle, registerDieId, "CcuReduceScatterKernel", | ||
| 210 | + reinterpret_cast<void *>(ops_hccl::CcuReduceScatterKernel), kernelArgs, 1, | ||
| 211 | + &resource.ccuKernels[groupIndex]); | ||
| 212 | + if (ccuResult != CCU_SUCCESS) { | ||
| 213 | + return ConvertCcuResult(ccuResult); | ||
| 214 | + } | ||
| 215 | + g_persistentKernelArgs.push_back(kernelArg); | ||
| 216 | + kernelArguments.push_back(std::move(kernelArg)); | ||
| 217 | + } | ||
| 218 | + | ||
| 219 | + if (needsCombineKernel) { | ||
| 220 | + for (uint32_t groupIndex = 0; groupIndex < groups.size(); ++groupIndex) { | ||
| 221 | + auto combineArg = std::make_shared<CcuCombineKernelArg>(); | ||
| 222 | + combineArg->channelCount = 1; | ||
| 223 | + combineArg->channels[0] = groups[groupIndex].channels[0]; | ||
| 224 | + const void *combineArgs[] = {combineArg.get()}; | ||
| 225 | + ccuResult = HcommCcuKernelRegister(instructionHandle, 0, "CcuCombineKernel", | ||
| 226 | + reinterpret_cast<void *>(ops_hccl::CcuCombineKernel), combineArgs, 1, | ||
| 227 | + &resource.ccuKernels[groups.size() + groupIndex]); | ||
| 228 | + if (ccuResult != CCU_SUCCESS) { | ||
| 229 | + return ConvertCcuResult(ccuResult); | ||
| 230 | + } | ||
| 231 | + g_persistentKernelArgs.push_back(combineArg); | ||
| 232 | + } | ||
| 233 | + } | ||
| 234 | + | ||
| 235 | + ccuResult = HcommCcuKernelRegisterEnd(instructionHandle); | ||
| 236 | + if (ccuResult != CCU_SUCCESS) { | ||
| 237 | + return ConvertCcuResult(ccuResult); | ||
| 238 | + } | ||
| 239 | + | ||
| 240 | + return HCCL_SUCCESS; | ||
| 241 | +} | ||
| 242 | +} // namespace | ||
| 243 | + | ||
| 244 | +HcclResult HcclReduceScatter(void *sendBuf, void *recvBuf, uint64_t recvCount, HcclDataType dataType, HcclReduceOp op, | ||
| 245 | + HcclComm comm, aclrtStream stream) | ||
| 246 | +{ | ||
| 247 | + CHK_PTR_NULL(sendBuf); | ||
| 248 | + CHK_PTR_NULL(recvBuf); | ||
| 249 | + CHK_PTR_NULL(comm); | ||
| 250 | + CHK_PTR_NULL(stream); | ||
| 251 | + | ||
| 252 | + if (dataType != HCCL_DATA_TYPE_FP32 || op != HCCL_REDUCE_SUM) { | ||
| 253 | + HCCL_ERROR("[ReduceScatter] Only float32 sum is supported"); | ||
| 254 | + return HCCL_E_NOT_SUPPORT; | ||
| 255 | + } | ||
| 256 | + | ||
| 257 | + // 构造算子参数 | ||
| 258 | + OpParam param; | ||
| 259 | + int tagLength = std::snprintf( | ||
| 260 | + param.tag, sizeof(param.tag), "%s", "hccl_custom_reducescatter_ccu_buffered_reduce"); | ||
| 261 | + if (tagLength <= 0 || static_cast<size_t>(tagLength) >= sizeof(param.tag)) { | ||
| 262 | + return HCCL_E_INTERNAL; | ||
| 263 | + } | ||
| 264 | + param.inputPtr = sendBuf; | ||
| 265 | + param.outputPtr = recvBuf; | ||
| 266 | + param.count = recvCount; | ||
| 267 | + param.dataType = dataType; | ||
| 268 | + param.opType = HcclCMDType::HCCL_CMD_REDUCE_SCATTER; | ||
| 269 | + param.reduceType = op; | ||
| 270 | + | ||
| 271 | + // 注册算子信息 | ||
| 272 | + HcclDfxOpInfo dfxInfo; | ||
| 273 | + char commName[COMM_INDENTIFIER_MAX_LENGTH]; | ||
| 274 | + CHK_RET(HcclGetCommName(comm, commName)); | ||
| 275 | + CHK_RET(HcclDfxRegOpInfoByCommId(commName, reinterpret_cast<void *>(&dfxInfo))); | ||
| 276 | + | ||
| 277 | + // ============================================== | ||
| 278 | + // STEP 1: 解析拓扑信息 | ||
| 279 | + // ============================================== | ||
| 280 | + CHK_RET(HcclGetRankId(comm, ¶m.myRank)); | ||
| 281 | + CHK_RET(HcclGetRankSize(comm, ¶m.rankSize)); | ||
| 282 | + | ||
| 283 | + // ============================================== | ||
| 284 | + // STEP 2: 创建资源 | ||
| 285 | + // ============================================== | ||
| 286 | + CommEngine ccuEngine = CommEngine::COMM_ENGINE_CCU; | ||
| 287 | + | ||
| 288 | + // 将用户传入的 stream 转换为 CCU 通信引擎中的 thread | ||
| 289 | + CHK_RET(HcclThreadAcquireWithStream(comm, ccuEngine, stream, 1, ¶m.cpuThread)); | ||
| 290 | + | ||
| 291 | + void *ctx = nullptr; | ||
| 292 | + uint64_t size = 0; | ||
| 293 | + if (HcclEngineCtxGet(comm, param.tag, ccuEngine, &ctx, &size) == HCCL_SUCCESS) { | ||
| 294 | + // CCU 资源已经存在,复用资源 | ||
| 295 | + HCCL_INFO("Engine context already exists"); | ||
| 296 | + param.resCtx = ctx; | ||
| 297 | + param.ctxSize = size; | ||
| 298 | + } else { | ||
| 299 | + // Device 资源不存在,资源构建 | ||
| 300 | + AlgResourceCtx resCtxHost{}; | ||
| 301 | + resCtxHost.ccuThread = param.cpuThread; | ||
| 302 | + | ||
| 303 | + // 从通信域获取 HCCL Buffer(Device上的内存,默认总大小400MB) | ||
| 304 | + void *cclBufferAddr; | ||
| 305 | + uint64_t cclBufferSize; | ||
| 306 | + CHK_RET(HcclGetHcclBuffer(comm, &cclBufferAddr, &cclBufferSize)); | ||
| 307 | + resCtxHost.localBuffer = CommBuffer{cclBufferAddr, cclBufferSize}; | ||
| 308 | + | ||
| 309 | + if (param.rankSize > 1) { | ||
| 310 | + std::vector<ChannelGroup> channelGroups; | ||
| 311 | + CHK_RET(AcquirePeerChannels(comm, ccuEngine, param.myRank, param.rankSize, channelGroups)); | ||
| 312 | + if (channelGroups.empty() || channelGroups.size() > MAX_DIE_NUM) { | ||
| 313 | + return HCCL_E_INTERNAL; | ||
| 314 | + } | ||
| 315 | + | ||
| 316 | + resCtxHost.threads.resize(channelGroups.size()); | ||
| 317 | + resCtxHost.threads[0] = param.cpuThread; | ||
| 318 | + if (channelGroups.size() > 1) { | ||
| 319 | + CHK_RET(HcclThreadAcquire( | ||
| 320 | + comm, ccuEngine, static_cast<uint32_t>(channelGroups.size() - 1), 1, &resCtxHost.threads[1])); | ||
| 321 | + } | ||
| 322 | + CHK_RET(RegisterCcuKernels(comm, param, channelGroups, resCtxHost)); | ||
| 323 | + } else { | ||
| 324 | + resCtxHost.threads = {param.cpuThread}; | ||
| 325 | + } | ||
| 326 | + | ||
| 327 | + // ============================================== | ||
| 328 | + // STEP 2.3: 申请通信引擎上下文 | ||
| 329 | + // ============================================== | ||
| 330 | + // 申请 CCU 通信引擎上下文,存放 AlgResourceCtx 信息 | ||
| 331 | + std::vector<char> seq = resCtxHost.Serialize(); | ||
| 332 | + uint64_t seqSize = seq.size(); | ||
| 333 | + param.ctxSize = seqSize; | ||
| 334 | + CHK_RET(HcclEngineCtxCreate(comm, param.tag, ccuEngine, param.ctxSize, ¶m.resCtx)); | ||
| 335 | + CHK_RET(HcclEngineCtxCopy(comm, ccuEngine, param.tag, seq.data(), seqSize, 0)); | ||
| 336 | + } | ||
| 337 | + | ||
| 338 | + // ============================================== | ||
| 339 | + // STEP 3: 下发 CCU Kernel | ||
| 340 | + // ============================================== | ||
| 341 | + CHK_RET(ops_hccl::ExecOp(param)); | ||
| 342 | + return HCCL_SUCCESS; | ||
| 343 | +} | ||
| @@ -0,0 +1,18 @@ | |||
| 1 | + # ----------------------------------------------------------------------------------------------------------- | ||
| 2 | +# Copyright (c) 2026 Huawei Technologies Co., Ltd. | ||
| 3 | +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of | ||
| 4 | +# CANN Open Software License Agreement Version 2.0 (the "License"). | ||
| 5 | +# Please refer to the License for details. You may not use this file except in compliance with the License. | ||
| 6 | +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, | ||
| 7 | +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. | ||
| 8 | +# See LICENSE in the root of the software repository for the full text of the License. | ||
| 9 | +# ----------------------------------------------------------------------------------------------------------- | ||
| 10 | + | ||
| 11 | +if(TARGET hccl) | ||
| 12 | + target_sources(hccl PRIVATE | ||
| 13 | + ${CMAKE_CURRENT_SOURCE_DIR}/ccu_kernel.cc | ||
| 14 | + ) | ||
| 15 | + target_include_directories(hccl PRIVATE | ||
| 16 | + ${CMAKE_CURRENT_SOURCE_DIR} | ||
| 17 | + ) | ||
| 18 | +endif() | ||
| @@ -0,0 +1,579 @@ | |||
| 1 | +/** | ||
| 2 | + * Copyright (c) 2026 Huawei Technologies Co., Ltd. | ||
| 3 | + * This program is free software, you can redistribute it and/or modify it under the terms and conditions of | ||
| 4 | + * CANN Open Software License Agreement Version 2.0 (the "License"). | ||
| 5 | + * Please refer to the License for details. You may not use this file except in compliance with the License. | ||
| 6 | + * THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, | ||
| 7 | + * INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. | ||
| 8 | + * See LICENSE in the root of the software repository for the full text of the License. | ||
| 9 | + */ | ||
| 10 | + | ||
| 11 | + | ||
| 12 | + | ||
| 13 | + | ||
| 14 | + | ||
| 15 | + | ||
| 16 | + | ||
| 17 | + | ||
| 18 | + | ||
| 19 | + | ||
| 20 | + | ||
| 21 | + do { \ | ||
| 22 | + CcuResult ccuResult = (call); \ | ||
| 23 | + if (ccuResult != CCU_SUCCESS) { \ | ||
| 24 | + return ccuResult; \ | ||
| 25 | + } \ | ||
| 26 | + } while (0) | ||
| 27 | + | ||
| 28 | + | ||
| 29 | +namespace ops_hccl { | ||
| 30 | +namespace { | ||
| 31 | + constexpr uint16_t INPUT_ADDRESS_RESOURCE_ID = 0; | ||
| 32 | + constexpr uint16_t INPUT_TOKEN_RESOURCE_ID = 1; | ||
| 33 | + constexpr uint16_t POST_SYNC_ID = 2; | ||
| 34 | + constexpr uint16_t CHANNEL_EVENT_INDEX = 0; | ||
| 35 | + constexpr uint16_t TRANSFER_EVENT_MASK = 1; | ||
| 36 | + constexpr uint32_t BUFFERED_REDUCE_MAX_SOURCES = 8; | ||
| 37 | + constexpr uint32_t BUFFERED_REDUCE_LOOP_COUNT = 16; | ||
| 38 | + constexpr uint32_t BUFFERED_REDUCE_INTERLEAVE = 8; | ||
| 39 | + constexpr uint64_t BUFFERED_REDUCE_SLICE_BYTES = 4ULL * 1024ULL; | ||
| 40 | + | ||
| 41 | + struct BufferedReduceGoSize { | ||
| 42 | + ccu::Variable enabled; | ||
| 43 | + ccu::Variable addressOffset; | ||
| 44 | + ccu::Variable loopIterations; | ||
| 45 | + ccu::Variable parallelConfig; | ||
| 46 | + ccu::Variable residualBytes; | ||
| 47 | + }; | ||
| 48 | + | ||
| 49 | + struct BufferedReduceResources { | ||
| 50 | + bool initialized = false; | ||
| 51 | + bool loopsCreated = false; | ||
| 52 | + ccu::Array<ccu::Event> completedEvents{0}; | ||
| 53 | + ccu::Array<ccu::CcuBuffer> buffers{0}; | ||
| 54 | + std::array<std::vector<ccu::LocalAddr>, 2> loopSources; | ||
| 55 | + ccu::LocalAddr loopDestination[2]; | ||
| 56 | + ccu::Variable loopLength[2]; | ||
| 57 | + ccu::Variable loopParameters[2]; | ||
| 58 | + std::array<std::unique_ptr<ccu::Func>, 2> bodies; | ||
| 59 | + std::array<std::unique_ptr<ccu::Loop>, 2> loops; | ||
| 60 | + }; | ||
| 61 | + | ||
| 62 | + struct ReduceScatterContext { | ||
| 63 | + const CcuReduceScatterKernelArg *arg; | ||
| 64 | + std::vector<ccu::Variable> remoteInputAddresses; | ||
| 65 | + std::vector<ccu::Variable> remoteInputTokens; | ||
| 66 | + ccu::Variable localInputAddress; | ||
| 67 | + ccu::Variable localInputToken; | ||
| 68 | + ccu::Variable destinationAddress; | ||
| 69 | + ccu::Variable destinationToken; | ||
| 70 | + ccu::Variable scratchAddress; | ||
| 71 | + ccu::Variable scratchToken; | ||
| 72 | + ccu::Variable sourceBaseOffset; | ||
| 73 | + ccu::Variable chunkBytes; | ||
| 74 | + ccu::Variable pipelineFirstBytes; | ||
| 75 | + ccu::Variable pipelineSecondBytes; | ||
| 76 | + BufferedReduceGoSize firstReduce; | ||
| 77 | + BufferedReduceGoSize secondReduce; | ||
| 78 | + BufferedReduceResources bufferedReduce; | ||
| 79 | + ccu::Event event; | ||
| 80 | + ccu::Event pipelineEvent; | ||
| 81 | + }; | ||
| 82 | + | ||
| 83 | + constexpr uint64_t BitMask(uint16_t bitCount) | ||
| 84 | + { | ||
| 85 | + return (uint64_t{1} << bitCount) - 1; | ||
| 86 | + } | ||
| 87 | + | ||
| 88 | + uint64_t PackLoopConfig(uint64_t addressOffset, uint64_t loopIterations) | ||
| 89 | + { | ||
| 90 | + return ((addressOffset & BitMask(32)) << 13) | (loopIterations & BitMask(13)); | ||
| 91 | + } | ||
| 92 | + | ||
| 93 | + uint64_t PackParallelConfig(uint64_t repeatCount, uint64_t repeatLoopIndex, uint64_t totalLoopCount) | ||
| 94 | + { | ||
| 95 | + return ((repeatCount & BitMask(7)) << 55) | ((repeatLoopIndex & BitMask(7)) << 48) | ||
| 96 | + | ((totalLoopCount & BitMask(7)) << 41); | ||
| 97 | + } | ||
| 98 | + | ||
| 99 | + uint64_t PackOffsetConfig(uint64_t addressOffset, uint64_t bufferOffset, uint64_t eventOffset) | ||
| 100 | + { | ||
| 101 | + return ((addressOffset & BitMask(32)) << 21) | ((bufferOffset & BitMask(11)) << 10) | ||
| 102 | + | (eventOffset & BitMask(10)); | ||
| 103 | + } | ||
| 104 | + | ||
| 105 | + CcuResult InitResources(ReduceScatterContext &context) | ||
| 106 | + { | ||
| 107 | + const auto *arg = context.arg; | ||
| 108 | + context.remoteInputAddresses.resize(arg->channelCount); | ||
| 109 | + context.remoteInputTokens.resize(arg->channelCount); | ||
| 110 | + | ||
| 111 | + for (uint32_t channel = 0; channel < arg->channelCount; ++channel) { | ||
| 112 | + context.remoteInputAddresses[channel] | ||
| 113 | + = ccu::GetResByChannel<ccu::Variable>(arg->channels[channel], INPUT_ADDRESS_RESOURCE_ID); | ||
| 114 | + context.remoteInputTokens[channel] | ||
| 115 | + = ccu::GetResByChannel<ccu::Variable>(arg->channels[channel], INPUT_TOKEN_RESOURCE_ID); | ||
| 116 | + } | ||
| 117 | + const uint32_t sourceCount = arg->channelCount + (arg->includeLocal ? 1U : 0U); | ||
| 118 | + if (sourceCount <= BUFFERED_REDUCE_MAX_SOURCES) { | ||
| 119 | + context.bufferedReduce.completedEvents = ccu::Array<ccu::Event>(BUFFERED_REDUCE_LOOP_COUNT); | ||
| 120 | + context.bufferedReduce.buffers = ccu::Array<ccu::CcuBuffer>( | ||
| 121 | + BUFFERED_REDUCE_LOOP_COUNT * BUFFERED_REDUCE_INTERLEAVE); | ||
| 122 | + context.bufferedReduce.initialized = true; | ||
| 123 | + } | ||
| 124 | + return CCU_SUCCESS; | ||
| 125 | + } | ||
| 126 | + | ||
| 127 | + CcuResult LoadBufferedReduceArgs(BufferedReduceGoSize &goSize, uint32_t &argumentIndex) | ||
| 128 | + { | ||
| 129 | + CCU_CHK_RET(ccu::LoadArg(goSize.enabled, argumentIndex++)); | ||
| 130 | + CCU_CHK_RET(ccu::LoadArg(goSize.addressOffset, argumentIndex++)); | ||
| 131 | + CCU_CHK_RET(ccu::LoadArg(goSize.loopIterations, argumentIndex++)); | ||
| 132 | + CCU_CHK_RET(ccu::LoadArg(goSize.parallelConfig, argumentIndex++)); | ||
| 133 | + CCU_CHK_RET(ccu::LoadArg(goSize.residualBytes, argumentIndex++)); | ||
| 134 | + return CCU_SUCCESS; | ||
| 135 | + } | ||
| 136 | + | ||
| 137 | + CcuResult LoadArguments(ReduceScatterContext &context) | ||
| 138 | + { | ||
| 139 | + uint32_t argumentIndex = 0; | ||
| 140 | + CCU_CHK_RET(ccu::LoadArg(context.localInputAddress, argumentIndex++)); | ||
| 141 | + CCU_CHK_RET(ccu::LoadArg(context.destinationAddress, argumentIndex++)); | ||
| 142 | + CCU_CHK_RET(ccu::LoadArg(context.localInputToken, argumentIndex++)); | ||
| 143 | + CCU_CHK_RET(ccu::LoadArg(context.destinationToken, argumentIndex++)); | ||
| 144 | + CCU_CHK_RET(ccu::LoadArg(context.scratchAddress, argumentIndex++)); | ||
| 145 | + CCU_CHK_RET(ccu::LoadArg(context.scratchToken, argumentIndex++)); | ||
| 146 | + CCU_CHK_RET(ccu::LoadArg(context.sourceBaseOffset, argumentIndex++)); | ||
| 147 | + CCU_CHK_RET(ccu::LoadArg(context.chunkBytes, argumentIndex++)); | ||
| 148 | + CCU_CHK_RET(ccu::LoadArg(context.pipelineFirstBytes, argumentIndex++)); | ||
| 149 | + CCU_CHK_RET(ccu::LoadArg(context.pipelineSecondBytes, argumentIndex++)); | ||
| 150 | + CCU_CHK_RET(LoadBufferedReduceArgs(context.firstReduce, argumentIndex)); | ||
| 151 | + CCU_CHK_RET(LoadBufferedReduceArgs(context.secondReduce, argumentIndex)); | ||
| 152 | + return CCU_SUCCESS; | ||
| 153 | + } | ||
| 154 | + | ||
| 155 | + CcuResult PreSync(ReduceScatterContext &context) | ||
| 156 | + { | ||
| 157 | + const auto *arg = context.arg; | ||
| 158 | + for (uint32_t channel = 0; channel < arg->channelCount; ++channel) { | ||
| 159 | + CCU_CHK_RET(ccu::WriteVariableWithNotify(arg->channels[channel], context.localInputAddress, | ||
| 160 | + INPUT_ADDRESS_RESOURCE_ID, CHANNEL_EVENT_INDEX, 1U << INPUT_ADDRESS_RESOURCE_ID)); | ||
| 161 | + CCU_CHK_RET(ccu::WriteVariableWithNotify(arg->channels[channel], context.localInputToken, | ||
| 162 | + INPUT_TOKEN_RESOURCE_ID, CHANNEL_EVENT_INDEX, 1U << INPUT_TOKEN_RESOURCE_ID)); | ||
| 163 | + } | ||
| 164 | + | ||
| 165 | + constexpr uint32_t preSyncMask = (1U << INPUT_ADDRESS_RESOURCE_ID) | (1U << INPUT_TOKEN_RESOURCE_ID); | ||
| 166 | + for (uint32_t channel = 0; channel < arg->channelCount; ++channel) { | ||
| 167 | + CCU_CHK_RET(ccu::NotifyWait(arg->channels[channel], CHANNEL_EVENT_INDEX, preSyncMask)); | ||
| 168 | + } | ||
| 169 | + return CCU_SUCCESS; | ||
| 170 | + } | ||
| 171 | + | ||
| 172 | + CcuResult PostSync(ReduceScatterContext &context) | ||
| 173 | + { | ||
| 174 | + const auto *arg = context.arg; | ||
| 175 | + for (uint32_t channel = 0; channel < arg->channelCount; ++channel) { | ||
| 176 | + CCU_CHK_RET(ccu::NotifyRecord(arg->channels[channel], CHANNEL_EVENT_INDEX, 1U << POST_SYNC_ID)); | ||
| 177 | + } | ||
| 178 | + for (uint32_t channel = 0; channel < arg->channelCount; ++channel) { | ||
| 179 | + CCU_CHK_RET(ccu::NotifyWait(arg->channels[channel], CHANNEL_EVENT_INDEX, 1U << POST_SYNC_ID)); | ||
| 180 | + } | ||
| 181 | + return CCU_SUCCESS; | ||
| 182 | + } | ||
| 183 | + | ||
| 184 | + CcuResult IssueSegmentReads(ReduceScatterContext &context, ccu::Variable scratchSegmentOffset, | ||
| 185 | + ccu::Variable sourceSegmentOffset, ccu::Variable segmentBytes, ccu::Event segmentEvent) | ||
| 186 | + { | ||
| 187 | + const auto *arg = context.arg; | ||
| 188 | + const uint32_t sourceCount = arg->channelCount + (arg->includeLocal ? 1U : 0U); | ||
| 189 | + ccu::LocalAddr scratchBase; | ||
| 190 | + scratchBase.addr = context.scratchAddress; | ||
| 191 | + scratchBase.addr += scratchSegmentOffset; | ||
| 192 | + for (uint32_t slot = 0; slot < arg->scratchSlotBase; ++slot) { | ||
| 193 | + scratchBase.addr += segmentBytes; | ||
| 194 | + } | ||
| 195 | + scratchBase.token = context.scratchToken; | ||
| 196 | + | ||
| 197 | + uint32_t sourceIndex = 0; | ||
| 198 | + uint32_t readEventMask = 0; | ||
| 199 | + for (uint32_t sourceRank = 0; sourceRank < arg->rankSize; ++sourceRank) { | ||
| 200 | + ccu::LocalAddr scratchSlot; | ||
| 201 | + scratchSlot.addr = scratchBase.addr; | ||
| 202 | + for (uint32_t slot = 0; slot < sourceIndex; ++slot) { | ||
| 203 | + scratchSlot.addr += segmentBytes; | ||
| 204 | + } | ||
| 205 | + scratchSlot.token = context.scratchToken; | ||
| 206 | + | ||
| 207 | + if (sourceRank == arg->rankId) { | ||
| 208 | + if (arg->includeLocal) { | ||
| 209 | + ccu::LocalAddr source; | ||
| 210 | + source.addr = context.localInputAddress; | ||
| 211 | + source.addr += context.sourceBaseOffset; | ||
| 212 | + source.addr += sourceSegmentOffset; | ||
| 213 | + source.token = context.localInputToken; | ||
| 214 | + | ||
| 215 | + const uint32_t eventMask = 1U << sourceIndex; | ||
| 216 | + CCU_CHK_RET(ccu::LocalCopy( | ||
| 217 | + scratchSlot, source, segmentBytes, segmentEvent, eventMask)); | ||
| 218 | + readEventMask |= eventMask; | ||
| 219 | + ++sourceIndex; | ||
| 220 | + } | ||
| 221 | + continue; | ||
| 222 | + } | ||
| 223 | + | ||
| 224 | + for (uint32_t channel = 0; channel < arg->channelCount; ++channel) { | ||
| 225 | + if (arg->remoteRanks[channel] != sourceRank) { | ||
| 226 | + continue; | ||
| 227 | + } | ||
| 228 | + | ||
| 229 | + ccu::RemoteAddr remoteSource; | ||
| 230 | + remoteSource.addr = context.remoteInputAddresses[channel]; | ||
| 231 | + remoteSource.addr += context.sourceBaseOffset; | ||
| 232 | + remoteSource.addr += sourceSegmentOffset; | ||
| 233 | + remoteSource.token = context.remoteInputTokens[channel]; | ||
| 234 | + | ||
| 235 | + const uint32_t eventMask = 1U << sourceIndex; | ||
| 236 | + CCU_CHK_RET(ccu::Read(arg->channels[channel], scratchSlot, remoteSource, segmentBytes, | ||
| 237 | + segmentEvent, eventMask)); | ||
| 238 | + readEventMask |= eventMask; | ||
| 239 | + ++sourceIndex; | ||
| 240 | + break; | ||
| 241 | + } | ||
| 242 | + } | ||
| 243 | + | ||
| 244 | + if (sourceIndex != sourceCount || readEventMask == 0) { | ||
| 245 | + return CCU_E_PARA; | ||
| 246 | + } | ||
| 247 | + return CCU_SUCCESS; | ||
| 248 | + } | ||
| 249 | + | ||
| 250 | + CcuResult CreateBufferedReduceLoops(ReduceScatterContext &context, uint32_t sourceCount) | ||
| 251 | + { | ||
| 252 | + auto &resources = context.bufferedReduce; | ||
| 253 | + if (!resources.initialized || sourceCount == 0 || sourceCount > BUFFERED_REDUCE_MAX_SOURCES) { | ||
| 254 | + return CCU_E_PARA; | ||
| 255 | + } | ||
| 256 | + if (resources.loopsCreated) { | ||
| 257 | + return CCU_SUCCESS; | ||
| 258 | + } | ||
| 259 | + | ||
| 260 | + for (uint32_t index = 0; index < 2; ++index) { | ||
| 261 | + resources.loopSources[index].resize(sourceCount); | ||
| 262 | + const uint32_t bufferBase = index * BUFFERED_REDUCE_INTERLEAVE; | ||
| 263 | + ccu::Event loopEvent = resources.completedEvents[index]; | ||
| 264 | + resources.bodies[index].reset(new ccu::Func( | ||
| 265 | + [&resources, index, bufferBase, loopEvent, sourceCount]() { | ||
| 266 | + for (uint32_t source = 0; source < sourceCount; ++source) { | ||
| 267 | + ccu::LocalCopy(resources.buffers[bufferBase + source], | ||
| 268 | + resources.loopSources[index][source], resources.loopLength[index], loopEvent, | ||
| 269 | + 1U << source); | ||
| 270 | + } | ||
| 271 | + ccu::EventWait(loopEvent, (1U << sourceCount) - 1U); | ||
| 272 | + | ||
| 273 | + if (sourceCount > 1) { | ||
| 274 | + std::vector<ccu::CcuBuffer> reduceBuffers; | ||
| 275 | + reduceBuffers.reserve(sourceCount); | ||
| 276 | + for (uint32_t source = 0; source < sourceCount; ++source) { | ||
| 277 | + reduceBuffers.push_back(resources.buffers[bufferBase + source]); | ||
| 278 | + } | ||
| 279 | + ccu::LocalReduce(reduceBuffers.data(), sourceCount, HCCL_DATA_TYPE_FP32, | ||
| 280 | + HCCL_DATA_TYPE_FP32, HCCL_REDUCE_SUM, resources.loopLength[index], loopEvent, | ||
| 281 | + TRANSFER_EVENT_MASK); | ||
| 282 | + ccu::EventWait(loopEvent, TRANSFER_EVENT_MASK); | ||
| 283 | + } | ||
| 284 | + | ||
| 285 | + ccu::LocalCopy(resources.loopDestination[index], resources.buffers[bufferBase], | ||
| 286 | + resources.loopLength[index], loopEvent, TRANSFER_EVENT_MASK); | ||
| 287 | + ccu::EventWait(loopEvent, TRANSFER_EVENT_MASK); | ||
| 288 | + })); | ||
| 289 | + resources.loops[index].reset( | ||
| 290 | + new ccu::Loop(resources.loopParameters[index], *resources.bodies[index])); | ||
| 291 | + } | ||
| 292 | + resources.loopsCreated = true; | ||
| 293 | + return CCU_SUCCESS; | ||
| 294 | + } | ||
| 295 | + | ||
| 296 | + CcuResult RunBufferedReduce(ReduceScatterContext &context, ccu::LocalAddr destination, | ||
| 297 | + std::vector<ccu::LocalAddr> sources, BufferedReduceGoSize &goSize) | ||
| 298 | + { | ||
| 299 | + auto &resources = context.bufferedReduce; | ||
| 300 | + ccu::Variable loopConfig; | ||
| 301 | + ccu::Variable sliceBytes; | ||
| 302 | + ccu::Variable parallelConfig; | ||
| 303 | + ccu::Variable offsetConfig; | ||
| 304 | + | ||
| 305 | + CCU_IF(goSize.loopIterations != 0) | ||
| 306 | + { | ||
| 307 | + loopConfig = PackLoopConfig( | ||
| 308 | + BUFFERED_REDUCE_SLICE_BYTES * BUFFERED_REDUCE_LOOP_COUNT, 0); | ||
| 309 | + loopConfig += goSize.loopIterations; | ||
| 310 | + sliceBytes = BUFFERED_REDUCE_SLICE_BYTES; | ||
| 311 | + for (uint32_t source = 0; source < sources.size(); ++source) { | ||
| 312 | + resources.loopSources[0][source].addr = sources[source].addr; | ||
| 313 | + resources.loopSources[0][source].token = sources[source].token; | ||
| 314 | + } | ||
| 315 | + resources.loopDestination[0].addr = destination.addr; | ||
| 316 | + resources.loopDestination[0].token = destination.token; | ||
| 317 | + resources.loopLength[0] = sliceBytes; | ||
| 318 | + parallelConfig = PackParallelConfig(BUFFERED_REDUCE_LOOP_COUNT - 1, 0, 1); | ||
| 319 | + offsetConfig = PackOffsetConfig( | ||
| 320 | + BUFFERED_REDUCE_SLICE_BYTES, BUFFERED_REDUCE_INTERLEAVE, 1); | ||
| 321 | + resources.loopParameters[0] = loopConfig; | ||
| 322 | + std::vector<ccu::Loop> loops{*resources.loops[0]}; | ||
| 323 | + ccu::LoopGroup group(parallelConfig, offsetConfig, BUFFERED_REDUCE_LOOP_COUNT, loops); | ||
| 324 | + } | ||
| 325 | + | ||
| 326 | + CCU_IF(goSize.parallelConfig != 0) | ||
| 327 | + { | ||
| 328 | + for (uint32_t source = 0; source < sources.size(); ++source) { | ||
| 329 | + sources[source].addr += goSize.addressOffset; | ||
| 330 | + } | ||
| 331 | + destination.addr += goSize.addressOffset; | ||
| 332 | + | ||
| 333 | + for (uint32_t source = 0; source < sources.size(); ++source) { | ||
| 334 | + resources.loopSources[0][source].addr = sources[source].addr; | ||
| 335 | + resources.loopSources[0][source].token = sources[source].token; | ||
| 336 | + } | ||
| 337 | + resources.loopDestination[0].addr = destination.addr; | ||
| 338 | + resources.loopDestination[0].token = destination.token; | ||
| 339 | + resources.loopLength[0] = goSize.residualBytes; | ||
| 340 | + | ||
| 341 | + for (uint32_t source = 0; source < sources.size(); ++source) { | ||
| 342 | + sources[source].addr += goSize.residualBytes; | ||
| 343 | + } | ||
| 344 | + destination.addr += goSize.residualBytes; | ||
| 345 | + sliceBytes = BUFFERED_REDUCE_SLICE_BYTES; | ||
| 346 | + for (uint32_t source = 0; source < sources.size(); ++source) { | ||
| 347 | + resources.loopSources[1][source].addr = sources[source].addr; | ||
| 348 | + resources.loopSources[1][source].token = sources[source].token; | ||
| 349 | + } | ||
| 350 | + resources.loopDestination[1].addr = destination.addr; | ||
| 351 | + resources.loopDestination[1].token = destination.token; | ||
| 352 | + resources.loopLength[1] = sliceBytes; | ||
| 353 | + | ||
| 354 | + resources.loopParameters[0] = PackLoopConfig(0, 1); | ||
| 355 | + resources.loopParameters[1] = PackLoopConfig(0, 1); | ||
| 356 | + offsetConfig = PackOffsetConfig( | ||
| 357 | + BUFFERED_REDUCE_SLICE_BYTES, BUFFERED_REDUCE_INTERLEAVE, 1); | ||
| 358 | + std::vector<ccu::Loop> loops{*resources.loops[0], *resources.loops[1]}; | ||
| 359 | + ccu::LoopGroup group( | ||
| 360 | + goSize.parallelConfig, offsetConfig, BUFFERED_REDUCE_LOOP_COUNT, loops); | ||
| 361 | + } | ||
| 362 | + return CCU_SUCCESS; | ||
| 363 | + } | ||
| 364 | + | ||
| 365 | + CcuResult BufferedReduceSegment(ReduceScatterContext &context, ccu::Variable scratchSegmentOffset, | ||
| 366 | + ccu::Variable sourceSegmentOffset, ccu::Variable segmentBytes, ccu::Event segmentEvent, | ||
| 367 | + BufferedReduceGoSize &goSize) | ||
| 368 | + { | ||
| 369 | + const auto *arg = context.arg; | ||
| 370 | + const uint32_t sourceCount = arg->channelCount + (arg->includeLocal ? 1U : 0U); | ||
| 371 | + const uint32_t readEventMask = (1U << sourceCount) - 1U; | ||
| 372 | + CCU_CHK_RET(ccu::EventWait(segmentEvent, readEventMask)); | ||
| 373 | + CCU_CHK_RET(CreateBufferedReduceLoops(context, sourceCount)); | ||
| 374 | + | ||
| 375 | + ccu::LocalAddr scratchBase; | ||
| 376 | + scratchBase.addr = context.scratchAddress; | ||
| 377 | + scratchBase.addr += scratchSegmentOffset; | ||
| 378 | + for (uint32_t slot = 0; slot < arg->scratchSlotBase; ++slot) { | ||
| 379 | + scratchBase.addr += segmentBytes; | ||
| 380 | + } | ||
| 381 | + scratchBase.token = context.scratchToken; | ||
| 382 | + | ||
| 383 | + std::vector<ccu::LocalAddr> sources; | ||
| 384 | + sources.reserve(sourceCount); | ||
| 385 | + for (uint32_t source = 0; source < sourceCount; ++source) { | ||
| 386 | + ccu::LocalAddr sourceAddress; | ||
| 387 | + sourceAddress.addr = scratchBase.addr; | ||
| 388 | + for (uint32_t slot = 0; slot < source; ++slot) { | ||
| 389 | + sourceAddress.addr += segmentBytes; | ||
| 390 | + } | ||
| 391 | + sourceAddress.token = context.scratchToken; | ||
| 392 | + sources.push_back(sourceAddress); | ||
| 393 | + } | ||
| 394 | + | ||
| 395 | + ccu::LocalAddr destination; | ||
| 396 | + if (arg->includeLocal) { | ||
| 397 | + destination.addr = context.destinationAddress; | ||
| 398 | + destination.addr += sourceSegmentOffset; | ||
| 399 | + destination.token = context.destinationToken; | ||
| 400 | + } else { | ||
| 401 | + destination.addr = scratchBase.addr; | ||
| 402 | + destination.token = scratchBase.token; | ||
| 403 | + } | ||
| 404 | + return RunBufferedReduce(context, destination, sources, goSize); | ||
| 405 | + } | ||
| 406 | + | ||
| 407 | + CcuResult TreeReduceSegment(ReduceScatterContext &context, ccu::Variable scratchSegmentOffset, | ||
| 408 | + ccu::Variable sourceSegmentOffset, ccu::Variable segmentBytes, ccu::Event segmentEvent) | ||
| 409 | + { | ||
| 410 | + const auto *arg = context.arg; | ||
| 411 | + const uint32_t sourceCount = arg->channelCount + (arg->includeLocal ? 1U : 0U); | ||
| 412 | + const uint32_t readEventMask = (1U << sourceCount) - 1U; | ||
| 413 | + CCU_CHK_RET(ccu::EventWait(segmentEvent, readEventMask)); | ||
| 414 | + | ||
| 415 | + ccu::LocalAddr scratchBase; | ||
| 416 | + scratchBase.addr = context.scratchAddress; | ||
| 417 | + scratchBase.addr += scratchSegmentOffset; | ||
| 418 | + for (uint32_t slot = 0; slot < arg->scratchSlotBase; ++slot) { | ||
| 419 | + scratchBase.addr += segmentBytes; | ||
| 420 | + } | ||
| 421 | + scratchBase.token = context.scratchToken; | ||
| 422 | + | ||
| 423 | + uint32_t remainingSources = sourceCount; | ||
| 424 | + while (remainingSources > 1) { | ||
| 425 | + const uint32_t reduceSources = remainingSources / 2; | ||
| 426 | + const uint32_t sourceSlot = remainingSources - reduceSources; | ||
| 427 | + | ||
| 428 | + ccu::LocalAddr reduceSource; | ||
| 429 | + reduceSource.addr = scratchBase.addr; | ||
| 430 | + for (uint32_t slot = 0; slot < sourceSlot; ++slot) { | ||
| 431 | + reduceSource.addr += segmentBytes; | ||
| 432 | + } | ||
| 433 | + reduceSource.token = context.scratchToken; | ||
| 434 | + | ||
| 435 | + ccu::Variable reduceBytes; | ||
| 436 | + reduceBytes = segmentBytes; | ||
| 437 | + for (uint32_t source = 1; source < reduceSources; ++source) { | ||
| 438 | + reduceBytes += segmentBytes; | ||
| 439 | + } | ||
| 440 | + | ||
| 441 | + CCU_CHK_RET(ccu::LocalReduce(scratchBase, reduceSource, reduceBytes, arg->dataType, arg->reduceOp, | ||
| 442 | + segmentEvent, TRANSFER_EVENT_MASK)); | ||
| 443 | + CCU_CHK_RET(ccu::EventWait(segmentEvent, TRANSFER_EVENT_MASK)); | ||
| 444 | + remainingSources -= reduceSources; | ||
| 445 | + } | ||
| 446 | + | ||
| 447 | + if (arg->includeLocal) { | ||
| 448 | + ccu::LocalAddr destination; | ||
| 449 | + destination.addr = context.destinationAddress; | ||
| 450 | + destination.addr += sourceSegmentOffset; | ||
| 451 | + destination.token = context.destinationToken; | ||
| 452 | + CCU_CHK_RET( | ||
| 453 | + ccu::LocalCopy(destination, scratchBase, segmentBytes, segmentEvent, TRANSFER_EVENT_MASK)); | ||
| 454 | + CCU_CHK_RET(ccu::EventWait(segmentEvent, TRANSFER_EVENT_MASK)); | ||
| 455 | + } | ||
| 456 | + return CCU_SUCCESS; | ||
| 457 | + } | ||
| 458 | + | ||
| 459 | + CcuResult ReduceSegment(ReduceScatterContext &context, ccu::Variable scratchSegmentOffset, | ||
| 460 | + ccu::Variable sourceSegmentOffset, ccu::Variable segmentBytes, ccu::Event segmentEvent, | ||
| 461 | + BufferedReduceGoSize &goSize) | ||
| 462 | + { | ||
| 463 | + const uint32_t sourceCount | ||
| 464 | + = context.arg->channelCount + (context.arg->includeLocal ? 1U : 0U); | ||
| 465 | + if (sourceCount > BUFFERED_REDUCE_MAX_SOURCES) { | ||
| 466 | + return TreeReduceSegment( | ||
| 467 | + context, scratchSegmentOffset, sourceSegmentOffset, segmentBytes, segmentEvent); | ||
| 468 | + } | ||
| 469 | + | ||
| 470 | + CCU_IF(goSize.enabled == 0) | ||
| 471 | + { | ||
| 472 | + CCU_CHK_RET(TreeReduceSegment( | ||
| 473 | + context, scratchSegmentOffset, sourceSegmentOffset, segmentBytes, segmentEvent)); | ||
| 474 | + } | ||
| 475 | + CCU_IF(goSize.enabled != 0) | ||
| 476 | + { | ||
| 477 | + CCU_CHK_RET(BufferedReduceSegment(context, scratchSegmentOffset, sourceSegmentOffset, | ||
| 478 | + segmentBytes, segmentEvent, goSize)); | ||
| 479 | + } | ||
| 480 | + return CCU_SUCCESS; | ||
| 481 | + } | ||
| 482 | + | ||
| 483 | + CcuResult ExecuteReduceScatter(ReduceScatterContext &context) | ||
| 484 | + { | ||
| 485 | + ccu::Variable zero; | ||
| 486 | + zero = 0; | ||
| 487 | + | ||
| 488 | + CCU_IF(context.pipelineFirstBytes == 0) | ||
| 489 | + { | ||
| 490 | + CCU_CHK_RET(IssueSegmentReads(context, zero, zero, context.chunkBytes, context.event)); | ||
| 491 | + CCU_CHK_RET(ReduceSegment( | ||
| 492 | + context, zero, zero, context.chunkBytes, context.event, context.firstReduce)); | ||
| 493 | + } | ||
| 494 | + | ||
| 495 | + CCU_IF(context.pipelineFirstBytes != 0) | ||
| 496 | + { | ||
| 497 | + ccu::Variable secondScratchOffset; | ||
| 498 | + secondScratchOffset = context.pipelineFirstBytes; | ||
| 499 | + for (uint32_t rank = 1; rank < context.arg->rankSize; ++rank) { | ||
| 500 | + secondScratchOffset += context.pipelineFirstBytes; | ||
| 501 | + } | ||
| 502 | + | ||
| 503 | + // Match the official mesh mem2mem schedule: issue the next read | ||
| 504 | + // batch before waiting for and reducing the current batch. | ||
| 505 | + CCU_CHK_RET(IssueSegmentReads( | ||
| 506 | + context, zero, zero, context.pipelineFirstBytes, context.event)); | ||
| 507 | + CCU_CHK_RET(IssueSegmentReads(context, secondScratchOffset, context.pipelineFirstBytes, | ||
| 508 | + context.pipelineSecondBytes, context.pipelineEvent)); | ||
| 509 | + CCU_CHK_RET(ReduceSegment( | ||
| 510 | + context, zero, zero, context.pipelineFirstBytes, context.event, context.firstReduce)); | ||
| 511 | + CCU_CHK_RET(ReduceSegment(context, secondScratchOffset, context.pipelineFirstBytes, | ||
| 512 | + context.pipelineSecondBytes, context.pipelineEvent, context.secondReduce)); | ||
| 513 | + } | ||
| 514 | + return CCU_SUCCESS; | ||
| 515 | + } | ||
| 516 | +} // namespace | ||
| 517 | + | ||
| 518 | +CcuResult CcuReduceScatterKernel(CcuKernelArg kernelArgument) | ||
| 519 | +{ | ||
| 520 | + auto *kernelArg = static_cast<CcuReduceScatterKernelArg *>(kernelArgument); | ||
| 521 | + if (kernelArg == nullptr || kernelArg->rankSize == 0 || kernelArg->rankSize > MAX_RANK_SIZE | ||
| 522 | + || kernelArg->rankId >= kernelArg->rankSize || kernelArg->channelCount >= kernelArg->rankSize | ||
| 523 | + || (!kernelArg->includeLocal && kernelArg->channelCount == 0) | ||
| 524 | + || kernelArg->scratchSlotBase >= kernelArg->rankSize | ||
| 525 | + || kernelArg->scratchSlotBase + kernelArg->channelCount + (kernelArg->includeLocal ? 1U : 0U) | ||
| 526 | + > kernelArg->rankSize) { | ||
| 527 | + return CCU_E_PARA; | ||
| 528 | + } | ||
| 529 | + | ||
| 530 | + ReduceScatterContext context; | ||
| 531 | + context.arg = kernelArg; | ||
| 532 | + | ||
| 533 | + CCU_CHK_RET(InitResources(context)); | ||
| 534 | + CCU_CHK_RET(LoadArguments(context)); | ||
| 535 | + CCU_CHK_RET(PreSync(context)); | ||
| 536 | + CCU_CHK_RET(ExecuteReduceScatter(context)); | ||
| 537 | + CCU_CHK_RET(PostSync(context)); | ||
| 538 | + | ||
| 539 | + return CCU_SUCCESS; | ||
| 540 | +} | ||
| 541 | + | ||
| 542 | +CcuResult CcuCombineKernel(CcuKernelArg kernelArgument) | ||
| 543 | +{ | ||
| 544 | + auto *kernelArg = static_cast<CcuCombineKernelArg *>(kernelArgument); | ||
| 545 | + if (kernelArg == nullptr || kernelArg->channelCount != 1) { | ||
| 546 | + return CCU_E_PARA; | ||
| 547 | + } | ||
| 548 | + | ||
| 549 | + // Associate this local-only kernel with the same die as the primary | ||
| 550 | + // reduction kernel. The resource itself is not used for data transfer. | ||
| 551 | + (void)ccu::GetResByChannel<ccu::Variable>(kernelArg->channels[0], INPUT_ADDRESS_RESOURCE_ID); | ||
| 552 | + | ||
| 553 | + ccu::Variable destinationAddress; | ||
| 554 | + ccu::Variable sourceAddress; | ||
| 555 | + ccu::Variable destinationToken; | ||
| 556 | + ccu::Variable sourceToken; | ||
| 557 | + ccu::Variable chunkBytes; | ||
| 558 | + uint32_t argumentIndex = 0; | ||
| 559 | + CCU_CHK_RET(ccu::LoadArg(destinationAddress, argumentIndex++)); | ||
| 560 | + CCU_CHK_RET(ccu::LoadArg(sourceAddress, argumentIndex++)); | ||
| 561 | + CCU_CHK_RET(ccu::LoadArg(destinationToken, argumentIndex++)); | ||
| 562 | + CCU_CHK_RET(ccu::LoadArg(sourceToken, argumentIndex++)); | ||
| 563 | + CCU_CHK_RET(ccu::LoadArg(chunkBytes, argumentIndex++)); | ||
| 564 | + | ||
| 565 | + ccu::LocalAddr destination; | ||
| 566 | + destination.addr = destinationAddress; | ||
| 567 | + destination.token = destinationToken; | ||
| 568 | + ccu::LocalAddr source; | ||
| 569 | + source.addr = sourceAddress; | ||
| 570 | + source.token = sourceToken; | ||
| 571 | + ccu::Event event; | ||
| 572 | + | ||
| 573 | + CCU_CHK_RET(ccu::LocalReduce(destination, source, chunkBytes, HCCL_DATA_TYPE_FP32, HCCL_REDUCE_SUM, event, | ||
| 574 | + TRANSFER_EVENT_MASK)); | ||
| 575 | + CCU_CHK_RET(ccu::EventWait(event, TRANSFER_EVENT_MASK)); | ||
| 576 | + return CCU_SUCCESS; | ||
| 577 | +} | ||
| 578 | + | ||
| 579 | +} // namespace ops_hccl | ||
| @@ -0,0 +1,27 @@ | |||
| 1 | +/** | ||
| 2 | + * Copyright (c) 2026 Huawei Technologies Co., Ltd. | ||
| 3 | + * This program is free software, you can redistribute it and/or modify it under the terms and conditions of | ||
| 4 | + * CANN Open Software License Agreement Version 2.0 (the "License"). | ||
| 5 | + * Please refer to the License for details. You may not use this file except in compliance with the License. | ||
| 6 | + * THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, | ||
| 7 | + * INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. | ||
| 8 | + * See LICENSE in the root of the software repository for the full text of the License. | ||
| 9 | + */ | ||
| 10 | + | ||
| 11 | + | ||
| 12 | + | ||
| 13 | + | ||
| 14 | + | ||
| 15 | + | ||
| 16 | + | ||
| 17 | + | ||
| 18 | +namespace ccu = ::AscendC::ccu; | ||
| 19 | + | ||
| 20 | +namespace ops_hccl { | ||
| 21 | + | ||
| 22 | +// CCU Kernel 函数 | ||
| 23 | +CcuResult CcuReduceScatterKernel(CcuKernelArg arg); | ||
| 24 | +CcuResult CcuCombineKernel(CcuKernelArg arg); | ||
| 25 | +} // namespace ops_hccl | ||
| 26 | + | ||
| 27 | + | ||
A03_university/CANN-HCCL-GHM-Competition-2026/preliminary/submissions/南方科技大学-超能陆战队/.clang-format+106-0
| @@ -0,0 +1,106 @@ | |||
| 1 | +# ----------------------------------------------------------------------------------------------------------- | ||
| 2 | +# Copyright (c) 2026 Huawei Technologies Co., Ltd. | ||
| 3 | +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of | ||
| 4 | +# CANN Open Software License Agreement Version 2.0 (the "License"). | ||
| 5 | +# Please refer to the License for details. You may not use this file except in compliance with the License. | ||
| 6 | +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, | ||
| 7 | +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. | ||
| 8 | +# See LICENSE in the root of the software repository for the full text of the License. | ||
| 9 | +# ----------------------------------------------------------------------------------------------------------- | ||
| 10 | + | ||
| 11 | +Language: Cpp | ||
| 12 | +Standard: c++17 | ||
| 13 | +TabWidth: 4 | ||
| 14 | +UseTab: Never | ||
| 15 | +UseCRLF: false | ||
| 16 | +AccessModifierOffset: -4 | ||
| 17 | +AlignConsecutiveAssignments: false | ||
| 18 | +AlignConsecutiveDeclarations: false | ||
| 19 | +AlignEscapedNewlines: DontAlign | ||
| 20 | +AlignOperands: true | ||
| 21 | +AlignTrailingComments: true | ||
| 22 | +AllowAllArgumentsOnNextLine: true | ||
| 23 | +AllowShortLambdasOnASingleLine: Empty | ||
| 24 | +AllowAllParametersOfDeclarationOnNextLine: false | ||
| 25 | +AllowShortBlocksOnASingleLine: false | ||
| 26 | +AllowShortCaseLabelsOnASingleLine: false | ||
| 27 | +AllowShortFunctionsOnASingleLine: false | ||
| 28 | +AllowShortIfStatementsOnASingleLine: false | ||
| 29 | +AllowShortLoopsOnASingleLine: false | ||
| 30 | +AlwaysBreakAfterDefinitionReturnType: None | ||
| 31 | +AlwaysBreakBeforeMultilineStrings: false | ||
| 32 | +AlwaysBreakTemplateDeclarations: MultiLine | ||
| 33 | +BinPackArguments: true | ||
| 34 | +BinPackParameters: true | ||
| 35 | +AlignAfterOpenBracket: DontAlign | ||
| 36 | + | ||
| 37 | +BraceWrapping: | ||
| 38 | + AfterCaseLabel: false | ||
| 39 | + AfterClass: false | ||
| 40 | + AfterControlStatement: Never | ||
| 41 | + AfterEnum: false | ||
| 42 | + AfterFunction: true | ||
| 43 | + AfterNamespace: false | ||
| 44 | + AfterStruct: false | ||
| 45 | + AfterUnion: false | ||
| 46 | + AfterExternBlock: false | ||
| 47 | + BeforeCatch: false | ||
| 48 | + BeforeElse: false | ||
| 49 | + IndentBraces: false | ||
| 50 | + SplitEmptyFunction: true | ||
| 51 | + SplitEmptyRecord: true | ||
| 52 | + SplitEmptyNamespace: true | ||
| 53 | + | ||
| 54 | +BreakBeforeBinaryOperators: All | ||
| 55 | +BreakBeforeBraces: Custom | ||
| 56 | +BreakBeforeTernaryOperators: true | ||
| 57 | +BreakConstructorInitializersBeforeComma: false | ||
| 58 | +BreakInheritanceList: AfterColon | ||
| 59 | +ColumnLimit: 120 | ||
| 60 | +CommentPragmas: '^ IWYU pragma:' | ||
| 61 | +PackConstructorInitializers: CurrentLine | ||
| 62 | +ConstructorInitializerIndentWidth: 4 | ||
| 63 | +ContinuationIndentWidth: 4 | ||
| 64 | +Cpp11BracedListStyle: true | ||
| 65 | +DerivePointerAlignment: false | ||
| 66 | +DisableFormat: false | ||
| 67 | +ExperimentalAutoDetectBinPacking: false | ||
| 68 | +ForEachMacros: [ foreach ] | ||
| 69 | + | ||
| 70 | +IncludeCategories: | ||
| 71 | + - Regex: '^<' | ||
| 72 | + Priority: 3 | ||
| 73 | + - Regex: '^"hccl' | ||
| 74 | + Priority: 2 | ||
| 75 | + - Regex: '.*' | ||
| 76 | + Priority: 1 | ||
| 77 | + | ||
| 78 | +IndentCaseLabels: true | ||
| 79 | +IndentWidth: 4 | ||
| 80 | +IndentWrappedFunctionNames: false | ||
| 81 | +KeepEmptyLinesAtTheStartOfBlocks: false | ||
| 82 | +MacroBlockBegin: '' | ||
| 83 | +MacroBlockEnd: '' | ||
| 84 | +MaxEmptyLinesToKeep: 1 | ||
| 85 | +NamespaceIndentation: Inner | ||
| 86 | + | ||
| 87 | +PenaltyBreakBeforeFirstCallParameter: 19 | ||
| 88 | +PenaltyBreakComment: 300 | ||
| 89 | +PenaltyBreakFirstLessLess: 120 | ||
| 90 | +PenaltyBreakString: 1000 | ||
| 91 | +PenaltyExcessCharacter: 1000000 | ||
| 92 | +PenaltyReturnTypeOnItsOwnLine: 60 | ||
| 93 | + | ||
| 94 | +PointerAlignment: Right | ||
| 95 | +ReflowComments: true | ||
| 96 | +SortIncludes: false | ||
| 97 | +SpaceAfterCStyleCast: false | ||
| 98 | +SpaceBeforeAssignmentOperators: true | ||
| 99 | +SpaceBeforeParens: ControlStatements | ||
| 100 | +SpaceInEmptyParentheses: false | ||
| 101 | +SpacesBeforeTrailingComments: 1 | ||
| 102 | +SpacesInAngles: false | ||
| 103 | +SpacesInContainerLiterals: true | ||
| 104 | +SpacesInCStyleCastParentheses: false | ||
| 105 | +SpacesInParentheses: false | ||
| 106 | +SpacesInSquareBrackets: false | ||
A03_university/CANN-HCCL-GHM-Competition-2026/preliminary/submissions/南方科技大学-超能陆战队/.gitattributes+11-0
| @@ -0,0 +1,11 @@ | |||
| 1 | +# ----------------------------------------------------------------------------------------------------------- | ||
| 2 | +# Copyright (c) 2026 Huawei Technologies Co., Ltd. | ||
| 3 | +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of | ||
| 4 | +# CANN Open Software License Agreement Version 2.0 (the "License"). | ||
| 5 | +# Please refer to the License for details. You may not use this file except in compliance with the License. | ||
| 6 | +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, | ||
| 7 | +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. | ||
| 8 | +# See LICENSE in the root of the software repository for the full text of the License. | ||
| 9 | +# ----------------------------------------------------------------------------------------------------------- | ||
| 10 | + | ||
| 11 | +* text=auto eol=lf | ||
| @@ -0,0 +1,35 @@ | |||
| 1 | +# ----------------------------------------------------------------------------------------------------------- | ||
| 2 | +# Copyright (c) 2026 Huawei Technologies Co., Ltd. | ||
| 3 | +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of | ||
| 4 | +# CANN Open Software License Agreement Version 2.0 (the "License"). | ||
| 5 | +# Please refer to the License for details. You may not use this file except in compliance with the License. | ||
| 6 | +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, | ||
| 7 | +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. | ||
| 8 | +# See LICENSE in the root of the software repository for the full text of the License. | ||
| 9 | +# ----------------------------------------------------------------------------------------------------------- | ||
| 10 | + | ||
| 11 | +# IDE | ||
| 12 | +.vscode/ | ||
| 13 | +.idea/ | ||
| 14 | +.cache/ | ||
| 15 | +.DS_Store | ||
| 16 | +*~ | ||
| 17 | +*.swp | ||
| 18 | +*.swo | ||
| 19 | + | ||
| 20 | +# Build | ||
| 21 | +build/ | ||
| 22 | +build_ut/ | ||
| 23 | +third_party/ | ||
| 24 | +__pycache__/ | ||
| 25 | + | ||
| 26 | +# Agent | ||
| 27 | +.omo/ | ||
| 28 | +.claude/ | ||
| 29 | +.opencode/ | ||
| 30 | +CLAUDE.md | ||
| 31 | +AGENTS.md | ||
| 32 | + | ||
| 33 | +# Logs | ||
| 34 | +logs/ | ||
| 35 | +*.log | ||
A03_university/CANN-HCCL-GHM-Competition-2026/preliminary/submissions/南方科技大学-超能陆战队/CMakeLists.txt+47-0
| @@ -0,0 +1,47 @@ | |||
| 1 | +# ----------------------------------------------------------------------------------------------------------- | ||
| 2 | +# Copyright (c) 2026 Huawei Technologies Co., Ltd. | ||
| 3 | +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of | ||
| 4 | +# CANN Open Software License Agreement Version 2.0 (the "License"). | ||
| 5 | +# Please refer to the License for details. You may not use this file except in compliance with the License. | ||
| 6 | +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, | ||
| 7 | +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. | ||
| 8 | +# See LICENSE in the root of the software repository for the full text of the License. | ||
| 9 | +# ----------------------------------------------------------------------------------------------------------- | ||
| 10 | + | ||
| 11 | +cmake_minimum_required(VERSION 3.16.0) | ||
| 12 | + project(hccl_reduce_scatter_aicpu) | ||
| 13 | + | ||
| 14 | +message(STATUS "CMAKE_BUILD_TYPE=${CMAKE_BUILD_TYPE}") | ||
| 15 | +message(STATUS "ASCEND_CANN_PACKAGE_PATH=${ASCEND_CANN_PACKAGE_PATH}") | ||
| 16 | + | ||
| 17 | +set(HCC_TOOLCHAIN_DIR "${ASCEND_CANN_PACKAGE_PATH}/toolkit/toolchain/hcc") | ||
| 18 | +set(HCC_CXX_COMPILER "${HCC_TOOLCHAIN_DIR}/bin/aarch64-target-linux-gnu-g++") | ||
| 19 | +set(HCC_C_COMPILER "${HCC_TOOLCHAIN_DIR}/bin/aarch64-target-linux-gnu-gcc") | ||
| 20 | +set(HCC_C_AR "${HCC_TOOLCHAIN_DIR}/bin/aarch64-target-linux-gnu-ar") | ||
| 21 | + | ||
| 22 | +# 编译 Host 侧链接库 | ||
| 23 | +add_subdirectory(op_host) | ||
| 24 | + | ||
| 25 | +# 编译 Device 侧链接库 | ||
| 26 | +include(ExternalProject) | ||
| 27 | +ExternalProject_Add(hccl_device | ||
| 28 | + SOURCE_DIR ${CMAKE_SOURCE_DIR}/op_kernel_aicpu | ||
| 29 | + BINARY_DIR ${CMAKE_BINARY_DIR}/device_build | ||
| 30 | + CMAKE_ARGS | ||
| 31 | + -DTOOLCHAIN_DIR=${HCC_TOOLCHAIN_DIR} | ||
| 32 | + -DCMAKE_C_COMPILER=${HCC_C_COMPILER} | ||
| 33 | + -DCMAKE_CXX_COMPILER=${HCC_CXX_COMPILER} | ||
| 34 | + -DCMAKE_C_COMPILER_LAUNCHER=${CMAKE_C_COMPILER_LAUNCHER} | ||
| 35 | + -DCMAKE_CXX_COMPILER_LAUNCHER=${CMAKE_CXX_COMPILER_LAUNCHER} | ||
| 36 | + -DCMAKE_C_AR=${HCC_C_AR} | ||
| 37 | + -DCMAKE_BUILD_TYPE=${CMAKE_BUILD_TYPE} | ||
| 38 | + -DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX} | ||
| 39 | + -DASCEND_CANN_PACKAGE_PATH=${ASCEND_CANN_PACKAGE_PATH} | ||
| 40 | + INSTALL_COMMAND ${CMAKE_COMMAND} --install ${CMAKE_BINARY_DIR}/device_build --prefix ${CMAKE_INSTALL_PREFIX} | ||
| 41 | + BUILD_ALWAYS TRUE | ||
| 42 | +) | ||
| 43 | + | ||
| 44 | +# 安装 | ||
| 45 | +install(FILES include/hccl.h | ||
| 46 | + DESTINATION include | ||
| 47 | +) | ||
| @@ -0,0 +1,201 @@ | |||
| 1 | + Apache License | ||
| 2 | + Version 2.0, January 2004 | ||
| 3 | + http://www.apache.org/licenses/ | ||
| 4 | + | ||
| 5 | + TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION | ||
| 6 | + | ||
| 7 | + 1. Definitions. | ||
| 8 | + | ||
| 9 | + "License" shall mean the terms and conditions for use, reproduction, | ||
| 10 | + and distribution as defined by Sections 1 through 9 of this document. | ||
| 11 | + | ||
| 12 | + "Licensor" shall mean the copyright owner or entity authorized by | ||
| 13 | + the copyright owner that is granting the License. | ||
| 14 | + | ||
| 15 | + "Legal Entity" shall mean the union of the acting entity and all | ||
| 16 | + other entities that control, are controlled by, or are under common | ||
| 17 | + control with that entity. For the purposes of this definition, | ||
| 18 | + "control" means (i) the power, direct or indirect, to cause the | ||
| 19 | + direction or management of such entity, whether by contract or | ||
| 20 | + otherwise, or (ii) ownership of fifty percent (50%) or more of the | ||
| 21 | + outstanding shares, or (iii) beneficial ownership of such entity. | ||
| 22 | + | ||
| 23 | + "You" (or "Your") shall mean an individual or Legal Entity | ||
| 24 | + exercising permissions granted by this License. | ||
| 25 | + | ||
| 26 | + "Source" form shall mean the preferred form for making modifications, | ||
| 27 | + including but not limited to software source code, documentation | ||
| 28 | + source, and configuration files. | ||
| 29 | + | ||
| 30 | + "Object" form shall mean any form resulting from mechanical | ||
| 31 | + transformation or translation of a Source form, including but | ||
| 32 | + not limited to compiled object code, generated documentation, | ||
| 33 | + and conversions to other media types. | ||
| 34 | + | ||
| 35 | + "Work" shall mean the work of authorship, whether in Source or | ||
| 36 | + Object form, made available under the License, as indicated by a | ||
| 37 | + copyright notice that is included in or attached to the work | ||
| 38 | + (an example is provided in the Appendix below). | ||
| 39 | + | ||
| 40 | + "Derivative Works" shall mean any work, whether in Source or Object | ||
| 41 | + form, that is based on (or derived from) the Work and for which the | ||
| 42 | + editorial revisions, annotations, elaborations, or other modifications | ||
| 43 | + represent, as a whole, an original work of authorship. For the purposes | ||
| 44 | + of this License, Derivative Works shall not include works that remain | ||
| 45 | + separable from, or merely link (or bind by name) to the interfaces of, | ||
| 46 | + the Work and Derivative Works thereof. | ||
| 47 | + | ||
| 48 | + "Contribution" shall mean any work of authorship, including | ||
| 49 | + the original version of the Work and any modifications or additions | ||
| 50 | + to that Work or Derivative Works thereof, that is intentionally | ||
| 51 | + submitted to Licensor for inclusion in the Work by the copyright owner | ||
| 52 | + or by an individual or Legal Entity authorized to submit on behalf of | ||
| 53 | + the copyright owner. For the purposes of this definition, "submitted" | ||
| 54 | + means any form of electronic, verbal, or written communication sent | ||
| 55 | + to the Licensor or its representatives, including but not limited to | ||
| 56 | + communication on electronic mailing lists, source code control systems, | ||
| 57 | + and issue tracking systems that are managed by, or on behalf of, the | ||
| 58 | + Licensor for the purpose of discussing and improving the Work, but | ||
| 59 | + excluding communication that is conspicuously marked or otherwise | ||
| 60 | + designated in writing by the copyright owner as "Not a Contribution." | ||
| 61 | + | ||
| 62 | + "Contributor" shall mean Licensor and any individual or Legal Entity | ||
| 63 | + on behalf of whom a Contribution has been received by Licensor and | ||
| 64 | + subsequently incorporated within the Work. | ||
| 65 | + | ||
| 66 | + 2. Grant of Copyright License. Subject to the terms and conditions of | ||
| 67 | + this License, each Contributor hereby grants to You a perpetual, | ||
| 68 | + worldwide, non-exclusive, no-charge, royalty-free, irrevocable | ||
| 69 | + copyright license to reproduce, prepare Derivative Works of, | ||
| 70 | + publicly display, publicly perform, sublicense, and distribute the | ||
| 71 | + Work and such Derivative Works in Source or Object form. | ||
| 72 | + | ||
| 73 | + 3. Grant of Patent License. Subject to the terms and conditions of | ||
| 74 | + this License, each Contributor hereby grants to You a perpetual, | ||
| 75 | + worldwide, non-exclusive, no-charge, royalty-free, irrevocable | ||
| 76 | + (except as stated in this section) patent license to make, have made, | ||
| 77 | + use, offer to sell, sell, import, and otherwise transfer the Work, | ||
| 78 | + where such license applies only to those patent claims licensable | ||
| 79 | + by such Contributor that are necessarily infringed by their | ||
| 80 | + Contribution(s) alone or by combination of their Contribution(s) | ||
| 81 | + with the Work to which such Contribution(s) was submitted. If You | ||
| 82 | + institute patent litigation against any entity (including a | ||
| 83 | + cross-claim or counterclaim in a lawsuit) alleging that the Work | ||
| 84 | + or a Contribution incorporated within the Work constitutes direct | ||
| 85 | + or contributory patent infringement, then any patent licenses | ||
| 86 | + granted to You under this License for that Work shall terminate | ||
| 87 | + as of the date such litigation is filed. | ||
| 88 | + | ||
| 89 | + 4. Redistribution. You may reproduce and distribute copies of the | ||
| 90 | + Work or Derivative Works thereof in any medium, with or without | ||
| 91 | + modifications, and in Source or Object form, provided that You | ||
| 92 | + meet the following conditions: | ||
| 93 | + | ||
| 94 | + (a) You must give any other recipients of the Work or | ||
| 95 | + Derivative Works a copy of this License; and | ||
| 96 | + | ||
| 97 | + (b) You must cause any modified files to carry prominent notices | ||
| 98 | + stating that You changed the files; and | ||
| 99 | + | ||
| 100 | + (c) You must retain, in the Source form of any Derivative Works | ||
| 101 | + that You distribute, all copyright, patent, trademark, and | ||
| 102 | + attribution notices from the Source form of the Work, | ||
| 103 | + excluding those notices that do not pertain to any part of | ||
| 104 | + the Derivative Works; and | ||
| 105 | + | ||
| 106 | + (d) If the Work includes a "NOTICE" text file as part of its | ||
| 107 | + distribution, then any Derivative Works that You distribute must | ||
| 108 | + include a readable copy of the attribution notices contained | ||
| 109 | + within such NOTICE file, excluding those notices that do not | ||
| 110 | + pertain to any part of the Derivative Works, in at least one | ||
| 111 | + of the following places: within a NOTICE text file distributed | ||
| 112 | + as part of the Derivative Works; within the Source form or | ||
| 113 | + documentation, if provided along with the Derivative Works; or, | ||
| 114 | + within a display generated by the Derivative Works, if and | ||
| 115 | + wherever such third-party notices normally appear. The contents | ||
| 116 | + of the NOTICE file are for informational purposes only and | ||
| 117 | + do not modify the License. You may add Your own attribution | ||
| 118 | + notices within Derivative Works that You distribute, alongside | ||
| 119 | + or as an addendum to the NOTICE text from the Work, provided | ||
| 120 | + that such additional attribution notices cannot be construed | ||
| 121 | + as modifying the License. | ||
| 122 | + | ||
| 123 | + You may add Your own copyright statement to Your modifications and | ||
| 124 | + may provide additional or different license terms and conditions | ||
| 125 | + for use, reproduction, or distribution of Your modifications, or | ||
| 126 | + for any such Derivative Works as a whole, provided Your use, | ||
| 127 | + reproduction, and distribution of the Work otherwise complies with | ||
| 128 | + the conditions stated in this License. | ||
| 129 | + | ||
| 130 | + 5. Submission of Contributions. Unless You explicitly state otherwise, | ||
| 131 | + any Contribution intentionally submitted for inclusion in the Work | ||
| 132 | + by You to the Licensor shall be under the terms and conditions of | ||
| 133 | + this License, without any additional terms or conditions. | ||
| 134 | + Notwithstanding the above, nothing herein shall supersede or modify | ||
| 135 | + the terms of any separate license agreement you may have executed | ||
| 136 | + with Licensor regarding such Contributions. | ||
| 137 | + | ||
| 138 | + 6. Trademarks. This License does not grant permission to use the trade | ||
| 139 | + names, trademarks, service marks, or product names of the Licensor, | ||
| 140 | + except as required for reasonable and customary use in describing the | ||
| 141 | + origin of the Work and reproducing the content of the NOTICE file. | ||
| 142 | + | ||
| 143 | + 7. Disclaimer of Warranty. Unless required by applicable law or | ||
| 144 | + agreed to in writing, Licensor provides the Work (and each | ||
| 145 | + Contributor provides its Contributions) on an "AS IS" BASIS, | ||
| 146 | + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or | ||
| 147 | + implied, including, without limitation, any warranties or conditions | ||
| 148 | + of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A | ||
| 149 | + PARTICULAR PURPOSE. You are solely responsible for determining the | ||
| 150 | + appropriateness of using or redistributing the Work and assume any | ||
| 151 | + risks associated with Your exercise of permissions under this License. | ||
| 152 | + | ||
| 153 | + 8. Limitation of Liability. In no event and under no legal theory, | ||
| 154 | + whether in tort (including negligence), contract, or otherwise, | ||
| 155 | + unless required by applicable law (such as deliberate and grossly | ||
| 156 | + negligent acts) or agreed to in writing, shall any Contributor be | ||
| 157 | + liable to You for damages, including any direct, indirect, special, | ||
| 158 | + incidental, or consequential damages of any character arising as a | ||
| 159 | + result of this License or out of the use or inability to use the | ||
| 160 | + Work (including but not limited to damages for loss of goodwill, | ||
| 161 | + work stoppage, computer failure or malfunction, or any and all | ||
| 162 | + other commercial damages or losses), even if such Contributor | ||
| 163 | + has been advised of the possibility of such damages. | ||
| 164 | + | ||
| 165 | + 9. Accepting Warranty or Additional Liability. While redistributing | ||
| 166 | + the Work or Derivative Works thereof, You may choose to offer, | ||
| 167 | + and charge a fee for, acceptance of support, warranty, indemnity, | ||
| 168 | + or other liability obligations and/or rights consistent with this | ||
| 169 | + License. However, in accepting such obligations, You may act only | ||
| 170 | + on Your own behalf and on Your sole responsibility, not on behalf | ||
| 171 | + of any other Contributor, and only if You agree to indemnify, | ||
| 172 | + defend, and hold each Contributor harmless for any liability | ||
| 173 | + incurred by, or claims asserted against, such Contributor by reason | ||
| 174 | + of your accepting any such warranty or additional liability. | ||
| 175 | + | ||
| 176 | + END OF TERMS AND CONDITIONS | ||
| 177 | + | ||
| 178 | + APPENDIX: How to apply the Apache License to your work. | ||
| 179 | + | ||
| 180 | + To apply the Apache License to your work, attach the following | ||
| 181 | + boilerplate notice, with the fields enclosed by brackets "{}" | ||
| 182 | + replaced with your own identifying information. (Don't include | ||
| 183 | + the brackets!) The text should be enclosed in the appropriate | ||
| 184 | + comment syntax for the file format. We also recommend that a | ||
| 185 | + file or class name and description of purpose be included on the | ||
| 186 | + same "printed page" as the copyright notice for easier | ||
| 187 | + identification within third-party archives. | ||
| 188 | + | ||
| 189 | + Copyright {yyyy} {name of copyright owner} | ||
| 190 | + | ||
| 191 | + Licensed under the Apache License, Version 2.0 (the "License"); | ||
| 192 | + you may not use this file except in compliance with the License. | ||
| 193 | + You may obtain a copy of the License at | ||
| 194 | + | ||
| 195 | + http://www.apache.org/licenses/LICENSE-2.0 | ||
| 196 | + | ||
| 197 | + Unless required by applicable law or agreed to in writing, software | ||
| 198 | + distributed under the License is distributed on an "AS IS" BASIS, | ||
| 199 | + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. | ||
| 200 | + See the License for the specific language governing permissions and | ||
| 201 | + limitations under the License. | ||
| @@ -0,0 +1,65 @@ | |||
| 1 | +# ReduceScatter 集合通信算子 | ||
| 2 | + | ||
| 3 | +## 1. 项目介绍 | ||
| 4 | + | ||
| 5 | +``` | ||
| 6 | +├── CMakeLists.txt # 顶层 CMake 配置 | ||
| 7 | +├── build.sh # 构建脚本 | ||
| 8 | +├── .clang-format # 代码风格配置 | ||
| 9 | +├── include/ # 头文件目录 | ||
| 10 | +│ ├── hccl.h # 集合通信算子头文件 | ||
| 11 | +│ ├── common.h # 通用数据结构定义 | ||
| 12 | +│ ├── custom.h # ★ 选手编写:自定义数据结构定义 | ||
| 13 | +│ ├── log.h # 日志宏定义 | ||
| 14 | +│ └── binary_stream.h # 序列化类定义 | ||
| 15 | +├── op_host/ # Host侧代码目录 | ||
| 16 | +│ ├── reduce_scatter.cc # ★ 选手编写:Host侧资源申请逻辑 | ||
| 17 | +│ └── launch_aicpu_kernel.cc # AICPU Kernel加载与下发逻辑 | ||
| 18 | +└── op_kernel_aicpu/ # Device侧代码目录 | ||
| 19 | + ├── aicpu_kernel.cc # AICPU Kernel函数实现 | ||
| 20 | + └── exec_op.cc # ★ 选手编写:通信算法编排逻辑 | ||
| 21 | +``` | ||
| 22 | + | ||
| 23 | +> [!NOTE] 注意: | ||
| 24 | +> 算子工程中已提前预制好固有逻辑,选手仅允许修改 `custom.h`、`reduce_scatter.cc`、`exec_op.cc` 共 3 个文件内容。 | ||
| 25 | + | ||
| 26 | +## 2. 编译运行 | ||
| 27 | + | ||
| 28 | +### 2.1 安装 CANN-Toolkit 包 | ||
| 29 | + | ||
| 30 | +请单击[下载链接](https://ascend.devcloud.huaweicloud.com/artifactory/cann-run-mirror/software/legacy/20260701000328953/),根据产品型号和环境架构下载对应软件包。安装命令如下,更多指导参考《[CANN软件安装指南](https://www.hiascend.com/document/redirect/CannCommunityInstWizard)》。 | ||
| 31 | + | ||
| 32 | +```bash | ||
| 33 | +# 确保安装包具有可执行权限 | ||
| 34 | +chmod +x Ascend-cann-toolkit_9.1.0_linux-${arch}.run | ||
| 35 | +# 安装命令 | ||
| 36 | +./Ascend-cann-toolkit_9.1.0_linux-${arch}.run --full --install-path=${install_path} | ||
| 37 | +``` | ||
| 38 | + | ||
| 39 | +### 2.2 环境变量配置 | ||
| 40 | + | ||
| 41 | +按需选择合适的命令使环境变量生效。 | ||
| 42 | + | ||
| 43 | +```bash | ||
| 44 | +# 默认路径安装,以root用户为例(非root用户,将/usr/local替换为${HOME}) | ||
| 45 | +source /usr/local/Ascend/cann/set_env.sh | ||
| 46 | +# 指定路径安装 | ||
| 47 | +# source ${install_path}/cann/set_env.sh | ||
| 48 | +``` | ||
| 49 | + | ||
| 50 | +### 2.3 编译算子工程 | ||
| 51 | + | ||
| 52 | +```bash | ||
| 53 | +bash build.sh | ||
| 54 | + | ||
| 55 | +# 编译 Debug 版本,便于断点调试 | ||
| 56 | +bash build.sh --debug | ||
| 57 | +``` | ||
| 58 | + | ||
| 59 | +## 3. 代码格式 | ||
| 60 | + | ||
| 61 | +选手代码需符合 [.clang-format](.clang-format) 文件中的代码风格规范,可通过下列命令一键修改: | ||
| 62 | + | ||
| 63 | +```bash | ||
| 64 | +bash build.sh --format | ||
| 65 | +``` | ||
| @@ -0,0 +1,10 @@ | |||
| 1 | +# 超能陆战队——初赛代码归档 | ||
| 2 | + | ||
| 3 | +- 赛区:粤港澳赛区 | ||
| 4 | +- 学校:南方科技大学 | ||
| 5 | +- 队伍:超能陆战队 | ||
| 6 | +- 成员:Sami_Hui、Qu_、Alextory-Frank | ||
| 7 | +- 赛题:ReduceScatter(AICPU + TS) | ||
| 8 | +- 归档版本:`26746a8` | ||
| 9 | + | ||
| 10 | +本目录保留了初赛阶段实际开发的完整工程代码以及原始提交历史。实现针对两服务器拓扑完成了确定性 ReduceScatter,并对通信读取、切片和本地归约流程进行了优化。 | ||
| @@ -0,0 +1,101 @@ | |||
| 1 | +# ----------------------------------------------------------------------------------------------------------- | ||
| 2 | +# Copyright (c) 2026 Huawei Technologies Co., Ltd. | ||
| 3 | +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of | ||
| 4 | +# CANN Open Software License Agreement Version 2.0 (the "License"). | ||
| 5 | +# Please refer to the License for details. You may not use this file except in compliance with the License. | ||
| 6 | +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, | ||
| 7 | +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. | ||
| 8 | +# See LICENSE in the root of the software repository for the full text of the License. | ||
| 9 | +# ----------------------------------------------------------------------------------------------------------- | ||
| 10 | + | ||
| 11 | +set -e | ||
| 12 | + | ||
| 13 | +PROJECT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" | ||
| 14 | +BUILD_DIR="${PROJECT_DIR}/build" | ||
| 15 | +BUILD_TYPE="Release" | ||
| 16 | +ENABLE_FORMAT="OFF" | ||
| 17 | + | ||
| 18 | +CPU_NUM="$(nproc)" | ||
| 19 | +ASCEND_CANN_PACKAGE_PATH="/usr/local/Ascend/cann" | ||
| 20 | + | ||
| 21 | +usage() { | ||
| 22 | + cat <<EOF | ||
| 23 | +Usage: $0 [OPTIONS] | ||
| 24 | + | ||
| 25 | +Options: | ||
| 26 | + --debug 编译 Debug 版本 | ||
| 27 | + --format 格式化代码 | ||
| 28 | + -h, --help 显示帮助信息 | ||
| 29 | +EOF | ||
| 30 | + exit 0 | ||
| 31 | +} | ||
| 32 | + | ||
| 33 | +parse_args() { | ||
| 34 | + local opts | ||
| 35 | + opts=$(getopt -o h -l debug,format,help -- "$@") || usage | ||
| 36 | + eval set -- "${opts}" | ||
| 37 | + | ||
| 38 | + for arg; do | ||
| 39 | + case "${arg}" in | ||
| 40 | + --debug) BUILD_TYPE="Debug" ;; | ||
| 41 | + --format) ENABLE_FORMAT="ON" ;; | ||
| 42 | + -h|--help) usage ;; | ||
| 43 | + esac | ||
| 44 | + done | ||
| 45 | +} | ||
| 46 | + | ||
| 47 | +parse_cann_path() { | ||
| 48 | + if [[ -z "${ASCEND_HOME_PATH}" ]]; then | ||
| 49 | + printf "ERROR: ASCEND_HOME_PATH is not set.\n" >&2 | ||
| 50 | + printf "Please ensure CANN-Toolkit is properly installed and source environment variables by running:\n" >&2 | ||
| 51 | + printf " source /path/to/Ascend/cann/set_env.sh\n" >&2 | ||
| 52 | + exit 1 | ||
| 53 | + fi | ||
| 54 | + | ||
| 55 | + # 使用 local 避免污染全局(如需要全局可去掉 local) | ||
| 56 | + ASCEND_CANN_PACKAGE_PATH="${ASCEND_HOME_PATH}" | ||
| 57 | + return 0 | ||
| 58 | +} | ||
| 59 | + | ||
| 60 | +build() { | ||
| 61 | + # 创建构建目录 | ||
| 62 | + cd "${PROJECT_DIR}" | ||
| 63 | + mkdir -p "${BUILD_DIR}" | ||
| 64 | + | ||
| 65 | + # 配置 | ||
| 66 | + cmake -S . -B "${BUILD_DIR}" \ | ||
| 67 | + -DCMAKE_EXPORT_COMPILE_COMMANDS=ON \ | ||
| 68 | + -DCMAKE_BUILD_TYPE="${BUILD_TYPE}" \ | ||
| 69 | + -DCMAKE_INSTALL_PREFIX="${BUILD_DIR}" \ | ||
| 70 | + -DASCEND_CANN_PACKAGE_PATH="${ASCEND_CANN_PACKAGE_PATH}" | ||
| 71 | + | ||
| 72 | + # 编译 | ||
| 73 | + cmake --build "${BUILD_DIR}" -j${CPU_NUM} | ||
| 74 | + | ||
| 75 | + # 安装 | ||
| 76 | + cmake --install "${BUILD_DIR}" | ||
| 77 | +} | ||
| 78 | + | ||
| 79 | +format() { | ||
| 80 | + cd "${PROJECT_DIR}" | ||
| 81 | + find ./include -type f -regex '.*\.\(h\|hpp\|cpp\|cc\)$' -exec clang-format -i {} \; | ||
| 82 | + find ./op_host -type f -regex '.*\.\(h\|hpp\|cpp\|cc\)$' -exec clang-format -i {} \; | ||
| 83 | + find ./op_kernel_aicpu -type f -regex '.*\.\(h\|hpp\|cpp\|cc\)$' -exec clang-format -i {} \; | ||
| 84 | +} | ||
| 85 | + | ||
| 86 | +main() { | ||
| 87 | + # 解析参数 | ||
| 88 | + parse_args "$@" | ||
| 89 | + # 解析 CANN-Toolkit 路径 | ||
| 90 | + parse_cann_path | ||
| 91 | + | ||
| 92 | + if [[ "${ENABLE_FORMAT}" == "ON" ]]; then | ||
| 93 | + # 格式化代码 | ||
| 94 | + format | ||
| 95 | + else | ||
| 96 | + # 编译 | ||
| 97 | + build | ||
| 98 | + fi | ||
| 99 | +} | ||
| 100 | + | ||
| 101 | +main "$@" | ||