已合并
【2026 HCCL通信库创新大赛-粤港澳赛区】【超能陆战队】初赛及决赛代码归档 #931
【2026 HCCL通信库创新大赛-粤港澳赛区】【超能陆战队】初赛及决赛代码归档 #931
已合并
liangjinxuan-2026创建于 8月3日
共 41 个文件变更+4110-0
@@ -0,0 +1,106 @@
1+# -----------------------------------------------------------------------------------------------------------
2+# Copyright (c) 2026 Huawei Technologies Co., Ltd.
3+# This program is free software, you can redistribute it and/or modify it under the terms and conditions of
4+# CANN Open Software License Agreement Version 2.0 (the "License").
5+# Please refer to the License for details. You may not use this file except in compliance with the License.
6+# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
7+# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
8+# See LICENSE in the root of the software repository for the full text of the License.
9+# -----------------------------------------------------------------------------------------------------------
10+ 
11+Language: Cpp
12+Standard: c++17
13+TabWidth: 4
14+UseTab: Never
15+UseCRLF: false
16+AccessModifierOffset: -4
17+AlignConsecutiveAssignments: false
18+AlignConsecutiveDeclarations: false
19+AlignEscapedNewlines: DontAlign
20+AlignOperands: true
21+AlignTrailingComments: true
22+AllowAllArgumentsOnNextLine: true
23+AllowShortLambdasOnASingleLine: Empty
24+AllowAllParametersOfDeclarationOnNextLine: false
25+AllowShortBlocksOnASingleLine: false
26+AllowShortCaseLabelsOnASingleLine: false
27+AllowShortFunctionsOnASingleLine: false
28+AllowShortIfStatementsOnASingleLine: false
29+AllowShortLoopsOnASingleLine: false
30+AlwaysBreakAfterDefinitionReturnType: None
31+AlwaysBreakBeforeMultilineStrings: false
32+AlwaysBreakTemplateDeclarations: MultiLine
33+BinPackArguments: true
34+BinPackParameters: true
35+AlignAfterOpenBracket: DontAlign
36+ 
37+BraceWrapping:
38+ AfterCaseLabel: false
39+ AfterClass: false
40+ AfterControlStatement: Never
41+ AfterEnum: false
42+ AfterFunction: true
43+ AfterNamespace: false
44+ AfterStruct: false
45+ AfterUnion: false
46+ AfterExternBlock: false
47+ BeforeCatch: false
48+ BeforeElse: false
49+ IndentBraces: false
50+ SplitEmptyFunction: true
51+ SplitEmptyRecord: true
52+ SplitEmptyNamespace: true
53+ 
54+BreakBeforeBinaryOperators: All
55+BreakBeforeBraces: Custom
56+BreakBeforeTernaryOperators: true
57+BreakConstructorInitializersBeforeComma: false
58+BreakInheritanceList: AfterColon
59+ColumnLimit: 120
60+CommentPragmas: '^ IWYU pragma:'
61+PackConstructorInitializers: CurrentLine
62+ConstructorInitializerIndentWidth: 4
63+ContinuationIndentWidth: 4
64+Cpp11BracedListStyle: true
65+DerivePointerAlignment: false
66+DisableFormat: false
67+ExperimentalAutoDetectBinPacking: false
68+ForEachMacros: [ foreach ]
69+ 
70+IncludeCategories:
71+ - Regex: '^<'
72+ Priority: 3
73+ - Regex: '^"hccl'
74+ Priority: 2
75+ - Regex: '.*'
76+ Priority: 1
77+ 
78+IndentCaseLabels: true
79+IndentWidth: 4
80+IndentWrappedFunctionNames: false
81+KeepEmptyLinesAtTheStartOfBlocks: false
82+MacroBlockBegin: ''
83+MacroBlockEnd: ''
84+MaxEmptyLinesToKeep: 1
85+NamespaceIndentation: Inner
86+ 
87+PenaltyBreakBeforeFirstCallParameter: 19
88+PenaltyBreakComment: 300
89+PenaltyBreakFirstLessLess: 120
90+PenaltyBreakString: 1000
91+PenaltyExcessCharacter: 1000000
92+PenaltyReturnTypeOnItsOwnLine: 60
93+ 
94+PointerAlignment: Right
95+ReflowComments: true
96+SortIncludes: false
97+SpaceAfterCStyleCast: false
98+SpaceBeforeAssignmentOperators: true
99+SpaceBeforeParens: ControlStatements
100+SpaceInEmptyParentheses: false
101+SpacesBeforeTrailingComments: 1
102+SpacesInAngles: false
103+SpacesInContainerLiterals: true
104+SpacesInCStyleCastParentheses: false
105+SpacesInParentheses: false
106+SpacesInSquareBrackets: false
@@ -0,0 +1,11 @@
1+# -----------------------------------------------------------------------------------------------------------
2+# Copyright (c) 2026 Huawei Technologies Co., Ltd.
3+# This program is free software, you can redistribute it and/or modify it under the terms and conditions of
4+# CANN Open Software License Agreement Version 2.0 (the "License").
5+# Please refer to the License for details. You may not use this file except in compliance with the License.
6+# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
7+# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
8+# See LICENSE in the root of the software repository for the full text of the License.
9+# -----------------------------------------------------------------------------------------------------------
10+ 
11+* text=auto eol=lf
@@ -0,0 +1,35 @@
1+# -----------------------------------------------------------------------------------------------------------
2+# Copyright (c) 2026 Huawei Technologies Co., Ltd.
3+# This program is free software, you can redistribute it and/or modify it under the terms and conditions of
4+# CANN Open Software License Agreement Version 2.0 (the "License").
5+# Please refer to the License for details. You may not use this file except in compliance with the License.
6+# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
7+# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
8+# See LICENSE in the root of the software repository for the full text of the License.
9+# -----------------------------------------------------------------------------------------------------------
10+ 
11+# IDE
12+.vscode/
13+.idea/
14+.cache/
15+.DS_Store
16+*~
17+*.swp
18+*.swo
19+ 
20+# Build
21+build/
22+build_ut/
23+third_party/
24+__pycache__/
25+ 
26+# Agent
27+.omo/
28+.claude/
29+.opencode/
30+CLAUDE.md
31+AGENTS.md
32+ 
33+# Logs
34+logs/
35+*.log
@@ -0,0 +1,24 @@
1+# -----------------------------------------------------------------------------------------------------------
2+# Copyright (c) 2026 Huawei Technologies Co., Ltd.
3+# This program is free software, you can redistribute it and/or modify it under the terms and conditions of
4+# CANN Open Software License Agreement Version 2.0 (the "License").
5+# Please refer to the License for details. You may not use this file except in compliance with the License.
6+# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
7+# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
8+# See LICENSE in the root of the software repository for the full text of the License.
9+# -----------------------------------------------------------------------------------------------------------
10+ 
11+cmake_minimum_required(VERSION 3.16.0)
12+project(hccl_reduce_scatter)
13+ 
14+message(STATUS "CMAKE_BUILD_TYPE=${CMAKE_BUILD_TYPE}")
15+message(STATUS "ASCEND_CANN_PACKAGE_PATH=${ASCEND_CANN_PACKAGE_PATH}")
16+ 
17+# 编译 Host 侧链接库
18+add_subdirectory(op_host)
19+add_subdirectory(op_kernel_ccu)
20+ 
21+# 安装
22+install(FILES include/hccl.h
23+ DESTINATION include
24+)
@@ -0,0 +1,201 @@
1+ Apache License
2+ Version 2.0, January 2004
3+ http://www.apache.org/licenses/
4+ 
5+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
6+ 
7+ 1. Definitions.
8+ 
9+ "License" shall mean the terms and conditions for use, reproduction,
10+ and distribution as defined by Sections 1 through 9 of this document.
11+ 
12+ "Licensor" shall mean the copyright owner or entity authorized by
13+ the copyright owner that is granting the License.
14+ 
15+ "Legal Entity" shall mean the union of the acting entity and all
16+ other entities that control, are controlled by, or are under common
17+ control with that entity. For the purposes of this definition,
18+ "control" means (i) the power, direct or indirect, to cause the
19+ direction or management of such entity, whether by contract or
20+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
21+ outstanding shares, or (iii) beneficial ownership of such entity.
22+ 
23+ "You" (or "Your") shall mean an individual or Legal Entity
24+ exercising permissions granted by this License.
25+ 
26+ "Source" form shall mean the preferred form for making modifications,
27+ including but not limited to software source code, documentation
28+ source, and configuration files.
29+ 
30+ "Object" form shall mean any form resulting from mechanical
31+ transformation or translation of a Source form, including but
32+ not limited to compiled object code, generated documentation,
33+ and conversions to other media types.
34+ 
35+ "Work" shall mean the work of authorship, whether in Source or
36+ Object form, made available under the License, as indicated by a
37+ copyright notice that is included in or attached to the work
38+ (an example is provided in the Appendix below).
39+ 
40+ "Derivative Works" shall mean any work, whether in Source or Object
41+ form, that is based on (or derived from) the Work and for which the
42+ editorial revisions, annotations, elaborations, or other modifications
43+ represent, as a whole, an original work of authorship. For the purposes
44+ of this License, Derivative Works shall not include works that remain
45+ separable from, or merely link (or bind by name) to the interfaces of,
46+ the Work and Derivative Works thereof.
47+ 
48+ "Contribution" shall mean any work of authorship, including
49+ the original version of the Work and any modifications or additions
50+ to that Work or Derivative Works thereof, that is intentionally
51+ submitted to Licensor for inclusion in the Work by the copyright owner
52+ or by an individual or Legal Entity authorized to submit on behalf of
53+ the copyright owner. For the purposes of this definition, "submitted"
54+ means any form of electronic, verbal, or written communication sent
55+ to the Licensor or its representatives, including but not limited to
56+ communication on electronic mailing lists, source code control systems,
57+ and issue tracking systems that are managed by, or on behalf of, the
58+ Licensor for the purpose of discussing and improving the Work, but
59+ excluding communication that is conspicuously marked or otherwise
60+ designated in writing by the copyright owner as "Not a Contribution."
61+ 
62+ "Contributor" shall mean Licensor and any individual or Legal Entity
63+ on behalf of whom a Contribution has been received by Licensor and
64+ subsequently incorporated within the Work.
65+ 
66+ 2. Grant of Copyright License. Subject to the terms and conditions of
67+ this License, each Contributor hereby grants to You a perpetual,
68+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
69+ copyright license to reproduce, prepare Derivative Works of,
70+ publicly display, publicly perform, sublicense, and distribute the
71+ Work and such Derivative Works in Source or Object form.
72+ 
73+ 3. Grant of Patent License. Subject to the terms and conditions of
74+ this License, each Contributor hereby grants to You a perpetual,
75+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
76+ (except as stated in this section) patent license to make, have made,
77+ use, offer to sell, sell, import, and otherwise transfer the Work,
78+ where such license applies only to those patent claims licensable
79+ by such Contributor that are necessarily infringed by their
80+ Contribution(s) alone or by combination of their Contribution(s)
81+ with the Work to which such Contribution(s) was submitted. If You
82+ institute patent litigation against any entity (including a
83+ cross-claim or counterclaim in a lawsuit) alleging that the Work
84+ or a Contribution incorporated within the Work constitutes direct
85+ or contributory patent infringement, then any patent licenses
86+ granted to You under this License for that Work shall terminate
87+ as of the date such litigation is filed.
88+ 
89+ 4. Redistribution. You may reproduce and distribute copies of the
90+ Work or Derivative Works thereof in any medium, with or without
91+ modifications, and in Source or Object form, provided that You
92+ meet the following conditions:
93+ 
94+ (a) You must give any other recipients of the Work or
95+ Derivative Works a copy of this License; and
96+ 
97+ (b) You must cause any modified files to carry prominent notices
98+ stating that You changed the files; and
99+ 
100+ (c) You must retain, in the Source form of any Derivative Works
101+ that You distribute, all copyright, patent, trademark, and
102+ attribution notices from the Source form of the Work,
103+ excluding those notices that do not pertain to any part of
104+ the Derivative Works; and
105+ 
106+ (d) If the Work includes a "NOTICE" text file as part of its
107+ distribution, then any Derivative Works that You distribute must
108+ include a readable copy of the attribution notices contained
109+ within such NOTICE file, excluding those notices that do not
110+ pertain to any part of the Derivative Works, in at least one
111+ of the following places: within a NOTICE text file distributed
112+ as part of the Derivative Works; within the Source form or
113+ documentation, if provided along with the Derivative Works; or,
114+ within a display generated by the Derivative Works, if and
115+ wherever such third-party notices normally appear. The contents
116+ of the NOTICE file are for informational purposes only and
117+ do not modify the License. You may add Your own attribution
118+ notices within Derivative Works that You distribute, alongside
119+ or as an addendum to the NOTICE text from the Work, provided
120+ that such additional attribution notices cannot be construed
121+ as modifying the License.
122+ 
123+ You may add Your own copyright statement to Your modifications and
124+ may provide additional or different license terms and conditions
125+ for use, reproduction, or distribution of Your modifications, or
126+ for any such Derivative Works as a whole, provided Your use,
127+ reproduction, and distribution of the Work otherwise complies with
128+ the conditions stated in this License.
129+ 
130+ 5. Submission of Contributions. Unless You explicitly state otherwise,
131+ any Contribution intentionally submitted for inclusion in the Work
132+ by You to the Licensor shall be under the terms and conditions of
133+ this License, without any additional terms or conditions.
134+ Notwithstanding the above, nothing herein shall supersede or modify
135+ the terms of any separate license agreement you may have executed
136+ with Licensor regarding such Contributions.
137+ 
138+ 6. Trademarks. This License does not grant permission to use the trade
139+ names, trademarks, service marks, or product names of the Licensor,
140+ except as required for reasonable and customary use in describing the
141+ origin of the Work and reproducing the content of the NOTICE file.
142+ 
143+ 7. Disclaimer of Warranty. Unless required by applicable law or
144+ agreed to in writing, Licensor provides the Work (and each
145+ Contributor provides its Contributions) on an "AS IS" BASIS,
146+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
147+ implied, including, without limitation, any warranties or conditions
148+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
149+ PARTICULAR PURPOSE. You are solely responsible for determining the
150+ appropriateness of using or redistributing the Work and assume any
151+ risks associated with Your exercise of permissions under this License.
152+ 
153+ 8. Limitation of Liability. In no event and under no legal theory,
154+ whether in tort (including negligence), contract, or otherwise,
155+ unless required by applicable law (such as deliberate and grossly
156+ negligent acts) or agreed to in writing, shall any Contributor be
157+ liable to You for damages, including any direct, indirect, special,
158+ incidental, or consequential damages of any character arising as a
159+ result of this License or out of the use or inability to use the
160+ Work (including but not limited to damages for loss of goodwill,
161+ work stoppage, computer failure or malfunction, or any and all
162+ other commercial damages or losses), even if such Contributor
163+ has been advised of the possibility of such damages.
164+ 
165+ 9. Accepting Warranty or Additional Liability. While redistributing
166+ the Work or Derivative Works thereof, You may choose to offer,
167+ and charge a fee for, acceptance of support, warranty, indemnity,
168+ or other liability obligations and/or rights consistent with this
169+ License. However, in accepting such obligations, You may act only
170+ on Your own behalf and on Your sole responsibility, not on behalf
171+ of any other Contributor, and only if You agree to indemnify,
172+ defend, and hold each Contributor harmless for any liability
173+ incurred by, or claims asserted against, such Contributor by reason
174+ of your accepting any such warranty or additional liability.
175+ 
176+ END OF TERMS AND CONDITIONS
177+ 
178+ APPENDIX: How to apply the Apache License to your work.
179+ 
180+ To apply the Apache License to your work, attach the following
181+ boilerplate notice, with the fields enclosed by brackets "{}"
182+ replaced with your own identifying information. (Don't include
183+ the brackets!) The text should be enclosed in the appropriate
184+ comment syntax for the file format. We also recommend that a
185+ file or class name and description of purpose be included on the
186+ same "printed page" as the copyright notice for easier
187+ identification within third-party archives.
188+ 
189+ Copyright {yyyy} {name of copyright owner}
190+ 
191+ Licensed under the Apache License, Version 2.0 (the "License");
192+ you may not use this file except in compliance with the License.
193+ You may obtain a copy of the License at
194+ 
195+ http://www.apache.org/licenses/LICENSE-2.0
196+ 
197+ Unless required by applicable law or agreed to in writing, software
198+ distributed under the License is distributed on an "AS IS" BASIS,
199+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
200+ See the License for the specific language governing permissions and
201+ limitations under the License.
@@ -0,0 +1,64 @@
1+# ReduceScatter 集合通信算子
2+ 
3+## 1. 项目介绍
4+ 
5+```
6+├── CMakeLists.txt # 顶层 CMake 配置
7+├── build.sh # 构建脚本
8+├── .clang-format # 代码风格配置
9+├── include/ # 头文件目录
10+│ ├── hccl.h # 集合通信算子头文件
11+│ ├── common.h # 通用数据结构定义
12+│ ├── custom.h # ★ 选手编写:自定义数据结构定义
13+│ ├── log.h # 日志宏定义
14+│ └── binary_stream.h # 序列化类定义
15+├── op_host/ # Host侧代码目录
16+│ ├── reduce_scatter.cc # ★ 选手编写:Host侧资源申请逻辑
17+│ └── exec_op.cc # ★ 选手编写:通信算法编排逻辑
18+└── op_kernel_ccu/ # CCU侧代码目录
19+ └── ccu_kernel.cc # ★ 选手编写:通信算法编排逻辑
20+```
21+ 
22+> [!NOTE] 注意:
23+> 算子工程中已提前预制好固有逻辑,选手仅允许修改 `custom.h`、`reduce_scatter.cc`、`exec_op.h`、`exec_op.cc`、`ccu_kernel.h`、`ccu_kernel.cc` 共 6 个文件内容。
24+ 
25+## 2. 编译运行
26+ 
27+### 2.1 安装 CANN-Toolkit 包
28+ 
29+请单击[下载链接](https://ascend.devcloud.huaweicloud.com/artifactory/cann-run-mirror/software/legacy/20260701000328953/),根据产品型号和环境架构下载对应软件包。安装命令如下,更多指导参考《[CANN软件安装指南](https://www.hiascend.com/document/redirect/CannCommunityInstWizard)》。
30+ 
31+```bash
32+# 确保安装包具有可执行权限
33+chmod +x Ascend-cann-toolkit_9.1.0_linux-${arch}.run
34+# 安装命令
35+./Ascend-cann-toolkit_9.1.0_linux-${arch}.run --full --install-path=${install_path}
36+```
37+ 
38+### 2.2 环境变量配置
39+ 
40+按需选择合适的命令使环境变量生效。
41+ 
42+```bash
43+# 默认路径安装,以root用户为例(非root用户,将/usr/local替换为${HOME})
44+source /usr/local/Ascend/cann/set_env.sh
45+# 指定路径安装
46+# source ${install_path}/cann/set_env.sh
47+```
48+ 
49+### 2.3 编译算子工程
50+ 
51+```bash
52+bash build.sh
53+ 
54+# 编译 Debug 版本,便于断点调试
55+bash build.sh --debug
56+```
57+ 
58+## 3. 代码格式
59+ 
60+选手代码需符合 [.clang-format](.clang-format) 文件中的代码风格规范,可通过下列命令一键修改:
61+ 
62+```bash
63+bash build.sh --format
64+```
@@ -0,0 +1,10 @@
1+# 超能陆战队——决赛代码归档
2+ 
3+- 赛区:粤港澳赛区
4+- 学校:南方科技大学
5+- 队伍:超能陆战队
6+- 成员:Sami_Hui、Qu_、Alextory-Frank
7+- 赛题:ReduceScatter(CCU)
8+- 归档版本:`802e278`
9+ 
10+本目录为决赛阶段最后提交的完整工程代码。实现覆盖 2×8、4×1、8+4 三种拓扑,采用双 Die 分组、分段读取流水线以及 CCU Buffer 本地归约,并保留确定性归约顺序。
@@ -0,0 +1,101 @@
1+# -----------------------------------------------------------------------------------------------------------
2+# Copyright (c) 2026 Huawei Technologies Co., Ltd.
3+# This program is free software, you can redistribute it and/or modify it under the terms and conditions of
4+# CANN Open Software License Agreement Version 2.0 (the "License").
5+# Please refer to the License for details. You may not use this file except in compliance with the License.
6+# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
7+# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
8+# See LICENSE in the root of the software repository for the full text of the License.
9+# -----------------------------------------------------------------------------------------------------------
10+ 
11+set -e
12+ 
13+PROJECT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
14+BUILD_DIR="${PROJECT_DIR}/build"
15+BUILD_TYPE="Release"
16+ENABLE_FORMAT="OFF"
17+ 
18+CPU_NUM="$(nproc)"
19+ASCEND_CANN_PACKAGE_PATH="/usr/local/Ascend/cann"
20+ 
21+usage() {
22+ cat <<EOF
23+Usage: $0 [OPTIONS]
24+ 
25+Options:
26+ --debug 编译 Debug 版本
27+ --format 格式化代码
28+ -h, --help 显示帮助信息
29+EOF
30+ exit 0
31+}
32+ 
33+parse_args() {
34+ local opts
35+ opts=$(getopt -o h -l debug,format,help -- "$@") || usage
36+ eval set -- "${opts}"
37+ 
38+ for arg; do
39+ case "${arg}" in
40+ --debug) BUILD_TYPE="Debug" ;;
41+ --format) ENABLE_FORMAT="ON" ;;
42+ -h|--help) usage ;;
43+ esac
44+ done
45+}
46+ 
47+parse_cann_path() {
48+ if [[ -z "${ASCEND_HOME_PATH}" ]]; then
49+ printf "ERROR: ASCEND_HOME_PATH is not set.\n" >&2
50+ printf "Please ensure CANN-Toolkit is properly installed and source environment variables by running:\n" >&2
51+ printf " source /path/to/Ascend/cann/set_env.sh\n" >&2
52+ exit 1
53+ fi
54+ 
55+ # 使用 local 避免污染全局(如需要全局可去掉 local)
56+ ASCEND_CANN_PACKAGE_PATH="${ASCEND_HOME_PATH}"
57+ return 0
58+}
59+ 
60+build() {
61+ # 创建构建目录
62+ cd "${PROJECT_DIR}"
63+ mkdir -p "${BUILD_DIR}"
64+ 
65+ # 配置
66+ cmake -S . -B "${BUILD_DIR}" \
67+ -DCMAKE_EXPORT_COMPILE_COMMANDS=ON \
68+ -DCMAKE_BUILD_TYPE="${BUILD_TYPE}" \
69+ -DCMAKE_INSTALL_PREFIX="${BUILD_DIR}" \
70+ -DASCEND_CANN_PACKAGE_PATH="${ASCEND_CANN_PACKAGE_PATH}"
71+ 
72+ # 编译
73+ cmake --build "${BUILD_DIR}" -j${CPU_NUM}
74+ 
75+ # 安装
76+ cmake --install "${BUILD_DIR}"
77+}
78+ 
79+format() {
80+ cd "${PROJECT_DIR}"
81+ find ./include -type f -regex '.*\.\(h\|hpp\|cpp\|cc\)$' -exec clang-format -i {} \;
82+ find ./op_host -type f -regex '.*\.\(h\|hpp\|cpp\|cc\)$' -exec clang-format -i {} \;
83+ find ./op_kernel_ccu -type f -regex '.*\.\(h\|hpp\|cpp\|cc\)$' -exec clang-format -i {} \;
84+}
85+ 
86+main() {
87+ # 解析参数
88+ parse_args "$@"
89+ # 解析 CANN-Toolkit 路径
90+ parse_cann_path
91+ 
92+ if [[ "${ENABLE_FORMAT}" == "ON" ]]; then
93+ # 格式化代码
94+ format
95+ else
96+ # 编译
97+ build
98+ fi
99+}
100+ 
101+main "$@"
@@ -0,0 +1,168 @@
1+/**
2+ * Copyright (c) 2026 Huawei Technologies Co., Ltd.
3+ * This program is free software, you can redistribute it and/or modify it under the terms and conditions of
4+ * CANN Open Software License Agreement Version 2.0 (the "License").
5+ * Please refer to the License for details. You may not use this file except in compliance with the License.
6+ * THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
7+ * INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
8+ * See LICENSE in the root of the software repository for the full text of the License.
9+ */
10+ 
11+#ifndef OPS_HCCL_BINARY_STREAM_H
12+#define OPS_HCCL_BINARY_STREAM_H
13+ 
14+#include <vector>
15+#include <cstdint>
16+#include <sstream>
17+#include <algorithm>
18+#include <map>
19+ 
20+#include "log.h"
21+ 
22+/**
23+ * @brief 序列化工具类
24+ */
25+class BinaryStream {
26+public:
27+ static constexpr std::ios_base::openmode DEFAULT_IOS_MODE = std::ios_base::in | std::ios_base::out;
28+ 
29+ explicit BinaryStream(std::ios_base::openmode mode = DEFAULT_IOS_MODE) : stream(mode | std::ios_base::binary) {};
30+ 
31+ explicit BinaryStream(std::vector<char> &buf, std::ios_base::openmode mode = DEFAULT_IOS_MODE)
32+ : stream(mode | std::ios_base::binary)
33+ {
34+ stream.rdbuf()->pubsetbuf(buf.data(), buf.size());
35+ }
36+ 
37+ template <typename T> BinaryStream &operator<<(const T &t)
38+ {
39+ stream.write(reinterpret_cast<const char *>(&t), sizeof(T));
40+ return *this;
41+ }
42+ 
43+ // 多级vector递归序列化
44+ template <typename T> BinaryStream &operator<<(const std::vector<T> &vec)
45+ {
46+ size_t size = vec.size();
47+ *this << size;
48+ for (const auto &elem : vec) {
49+ *this << elem;
50+ }
51+ return *this;
52+ }
53+ 
54+ // 对string的输入函数
55+ BinaryStream &operator<<(const std::string &s)
56+ {
57+ size_t size = s.size();
58+ stream.write(reinterpret_cast<const char *>(&size), sizeof(size_t)); // 写入长度
59+ stream.write(s.data(), size); // 写入字符数据
60+ return *this;
61+ }
62+ 
63+ template <typename T> BinaryStream &operator>>(T &t)
64+ {
65+ stream.read(reinterpret_cast<char *>(&t), sizeof(T));
66+ return *this;
67+ }
68+ 
69+ // 对string的读取函数
70+ BinaryStream &operator>>(std::string &s)
71+ {
72+ size_t size;
73+ stream.read(reinterpret_cast<char *>(&size), sizeof(size)); // 先从流中读取字符串长度
74+ s.resize(size); // 为string分配足够空间
75+ stream.read(&s[0], size); // 直接读取数据到string的缓冲区中,无需再分配内存
76+ return *this;
77+ }
78+ 
79+ // 多级vector递归反序列化
80+ template <typename T> BinaryStream &operator>>(std::vector<T> &vec)
81+ {
82+ size_t size;
83+ *this >> size;
84+ vec.resize(size);
85+ for (auto &elem : vec) {
86+ *this >> elem;
87+ }
88+ return *this;
89+ }
90+ 
91+ // map序列化
92+ template <typename T1, typename T2> BinaryStream &operator<<(const std::map<T1, T2> &m)
93+ {
94+ size_t size = m.size();
95+ *this << size;
96+ for (const auto &elem : m) {
97+ *this << elem.first;
98+ *this << elem.second;
99+ }
100+ return *this;
101+ }
102+ 
103+ // map反序列化
104+ template <typename T1, typename T2> BinaryStream &operator>>(std::map<T1, T2> &m)
105+ {
106+ size_t size;
107+ *this >> size;
108+ for (size_t i = 0; i < size; i++) {
109+ T1 key;
110+ *this >> key;
111+ T2 value;
112+ *this >> value;
113+ m[key] = value;
114+ }
115+ return *this;
116+ }
117+ 
118+ void Dump(std::vector<char> &vec)
119+ {
120+ std::for_each(std::istreambuf_iterator<char>(stream), std::istreambuf_iterator<char>(), [&vec](const char c) {
121+ vec.push_back(c);
122+ });
123+ }
124+ 
125+ void DumpWithRevert(std::vector<char> &vec)
126+ {
127+ std::streampos originalPos = stream.tellg(); // 保存原始位置
128+ std::for_each(std::istreambuf_iterator<char>(stream), std::istreambuf_iterator<char>(), [&vec](const char c) {
129+ vec.push_back(c);
130+ });
131+ stream.seekg(originalPos); // 恢复原始位置
132+ }
133+ 
134+ std::uint64_t GetSize()
135+ {
136+ return stream.str().size();
137+ }
138+ 
139+ std::string GetString()
140+ {
141+ return stream.str();
142+ }
143+ 
144+ std::string SplictStream(uint64_t &start, uint64_t &end)
145+ {
146+ std::string temp = stream.str();
147+ if (start >= temp.length()) {
148+ HCCL_ERROR("[SplictStream]start[%lu] is bigger than stream length[%lu]", start, temp.length());
149+ return "";
150+ }
151+ 
152+ // 截取子串
153+ std::string result = temp.substr(start, end - start);
154+ 
155+ // 返回新的 string
156+ return result;
157+ }
158+ 
159+ void Clear()
160+ {
161+ stream.clear();
162+ }
163+ 
164+private:
165+ std::stringstream stream;
166+};
167+ 
168+#endif // OPS_HCCL_BINARY_STREAM_H
@@ -0,0 +1,54 @@
1+/**
2+ * Copyright (c) 2026 Huawei Technologies Co., Ltd.
3+ * This program is free software, you can redistribute it and/or modify it under the terms and conditions of
4+ * CANN Open Software License Agreement Version 2.0 (the "License").
5+ * Please refer to the License for details. You may not use this file except in compliance with the License.
6+ * THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
7+ * INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
8+ * See LICENSE in the root of the software repository for the full text of the License.
9+ */
10+ 
11+#ifndef OPS_HCCL_COMMON_H
12+#define OPS_HCCL_COMMON_H
13+ 
14+#include <unordered_map>
15+ 
16+#include <hccl/hccl_types.h>
17+#include <hccl/hccl_res.h>
18+#include <hccl/hcomm_primitives.h>
19+#include <ccu/ccu_types.h>
20+#include <ccu/ccu_variable.hpp>
21+#include <ccu/ccu_event.hpp>
22+#include <ccu/ccu_primitives.hpp>
23+#include <acl/acl_rt.h>
24+ 
25+constexpr uint32_t NOTIFY_IDX_ACK = 0;
26+constexpr uint32_t NOTIFY_IDX_DATA_SIGNAL = 1;
27+constexpr uint32_t CUSTOM_TIMEOUT = 1800;
28+ 
29+constexpr uint32_t COMM_INDENTIFIER_MAX_LENGTH = 128;
30+constexpr uint32_t OP_NAME_LENGTH = 32;
31+constexpr uint32_t TAG_LENGTH = OP_NAME_LENGTH + COMM_INDENTIFIER_MAX_LENGTH;
32+constexpr uint32_t INVALID_VALUE_RANKID = 0xFFFFFFFF;
33+constexpr uint32_t MAX_DATA_SIZE = 256 * 1024 * 1024; // 单次通信的最大数据量,256MB
34+constexpr uint64_t MAX_RANK_SIZE = 16;
35+ 
36+struct OpParam {
37+ char tag[TAG_LENGTH];
38+ void *inputPtr = nullptr;
39+ void *outputPtr = nullptr;
40+ uint64_t count = 0;
41+ uint32_t root = 0;
42+ uint32_t myRank = INVALID_VALUE_RANKID;
43+ uint32_t rankSize = 0;
44+ HcclDataType dataType = HCCL_DATA_TYPE_RESERVED;
45+ HcclCMDType opType = HcclCMDType::HCCL_CMD_INVALID;
46+ HcclReduceOp reduceType = HcclReduceOp::HCCL_REDUCE_SUM;
47+ ThreadHandle cpuThread;
48+ void *resCtx = nullptr; ///< 通信引擎上下文中的资源信息,存放 AlgResourceCtx 序列化后的内容
49+ uint64_t ctxSize = 0;
50+};
51+ 
52+const std::unordered_map<HcclDataType, uint32_t> SIZE_TABLE = {{HCCL_DATA_TYPE_FP32, sizeof(float)}};
53+ 
54+#endif // OPS_HCCL_COMMON_H
@@ -0,0 +1,94 @@
1+/**
2+ * Copyright (c) 2026 Huawei Technologies Co., Ltd.
3+ * This program is free software, you can redistribute it and/or modify it under the terms and conditions of
4+ * CANN Open Software License Agreement Version 2.0 (the "License").
5+ * Please refer to the License for details. You may not use this file except in compliance with the License.
6+ * THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
7+ * INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
8+ * See LICENSE in the root of the software repository for the full text of the License.
9+ */
10+ 
11+#ifndef OPS_HCCL_CUSTOM_H
12+#define OPS_HCCL_CUSTOM_H
13+ 
14+#include <memory>
15+#include <vector>
16+#include <hccl/hccl_types.h>
17+#include <hccl/hccl_res.h>
18+ 
19+#include "binary_stream.h"
20+#include "common.h"
21+ 
22+typedef struct {
23+ void *addr;
24+ uint64_t size;
25+} CommBuffer;
26+ 
27+struct CcuKernelArgBase {
28+ ChannelHandle channels[MAX_RANK_SIZE];
29+ uint32_t channelCount;
30+};
31+ 
32+struct CcuReduceScatterKernelArg : public CcuKernelArgBase {
33+ uint32_t remoteRanks[MAX_RANK_SIZE];
34+ uint32_t rankSize;
35+ uint32_t rankId;
36+ uint32_t scratchSlotBase;
37+ bool includeLocal;
38+ HcclDataType dataType;
39+ HcclReduceOp reduceOp;
40+};
41+ 
42+struct CcuCombineKernelArg : public CcuKernelArgBase {};
43+ 
44+// ccu kernel register所需信息
45+struct CcuKernelInfo {
46+ // kernel名称
47+ char kernelFuncName[64];
48+ // kernel函数
49+ void *kernelFunc;
50+ // KernelArg实例指针
51+ void *kernelArg;
52+ 
53+private:
54+ std::shared_ptr<CcuKernelArgBase> kernelArgSmartPtr;
55+ 
56+public:
57+ template <typename T> void setKernelArg(std::shared_ptr<T> arg)
58+ {
59+ kernelArgSmartPtr = std::static_pointer_cast<CcuKernelArgBase>(arg);
60+ kernelArg = static_cast<void *>(arg.get());
61+ }
62+};
63+ 
64+struct AlgResourceCtx {
65+ ThreadHandle ccuThread; ///< CCU通信引擎上的thread资源
66+ CommBuffer localBuffer; ///< 本端HCCL通信内存
67+ std::vector<ThreadHandle> threads; ///< CCU通信引擎上的thread资源
68+ std::vector<CcuKernelHandle> ccuKernels;
69+ 
70+ // 序列化
71+ std::vector<char> Serialize()
72+ {
73+ BinaryStream binaryStream;
74+ binaryStream << ccuThread;
75+ binaryStream << localBuffer;
76+ binaryStream << threads;
77+ binaryStream << ccuKernels;
78+ std::vector<char> result;
79+ binaryStream.Dump(result);
80+ return result;
81+ }
82+ 
83+ // 反序列化
84+ void DeSerialize(std::vector<char> &data)
85+ {
86+ BinaryStream binaryStream(data);
87+ binaryStream >> ccuThread;
88+ binaryStream >> localBuffer;
89+ binaryStream >> threads;
90+ binaryStream >> ccuKernels;
91+ }
92+};
93+ 
94+#endif // OPS_HCCL_CUSTOM_H
@@ -0,0 +1,31 @@
1+/**
2+ * Copyright (c) 2026 Huawei Technologies Co., Ltd.
3+ * This program is free software, you can redistribute it and/or modify it under the terms and conditions of
4+ * CANN Open Software License Agreement Version 2.0 (the "License").
5+ * Please refer to the License for details. You may not use this file except in compliance with the License.
6+ * THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
7+ * INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
8+ * See LICENSE in the root of the software repository for the full text of the License.
9+ */
10+ 
11+#ifndef OPS_HCCL_H
12+#define OPS_HCCL_H
13+ 
14+#include <acl/acl.h>
15+#include <hccl/hccl_comm.h>
16+#include <hccl/hccl_res.h>
17+#include <hccl/hccl_types.h>
18+ 
19+#ifdef __cplusplus
20+extern "C" {
21+#endif
22+ 
23+/* ReduceScatter 算子入口 */
24+HcclResult HcclReduceScatter(void *sendBuf, void *recvBuf, uint64_t recvCount, HcclDataType dataType,
25+ HcclReduceOp op, HcclComm comm, aclrtStream stream);
26+ 
27+#ifdef __cplusplus
28+}
29+#endif
30+ 
31+#endif // OPS_HCCL_H
@@ -0,0 +1,140 @@
1+/**
2+ * Copyright (c) 2026 Huawei Technologies Co., Ltd.
3+ * This program is free software, you can redistribute it and/or modify it under the terms and conditions of
4+ * CANN Open Software License Agreement Version 2.0 (the "License").
5+ * Please refer to the License for details. You may not use this file except in compliance with the License.
6+ * THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
7+ * INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
8+ * See LICENSE in the root of the software repository for the full text of the License.
9+ */
10+ 
11+#ifndef OPS_HCCL_LOG_H
12+#define OPS_HCCL_LOG_H
13+ 
14+#include <cstdio>
15+#include <hccl/hccl_types.h>
16+#include <ccu/ccu_types.h>
17+ 
18+#ifndef LOG_LEVEL
19+#define LOG_LEVEL LOG_LEVEL_ERROR // 默认日志级别为:ERROR
20+#endif
21+ 
22+typedef enum {
23+ LOG_LEVEL_DEBUG = 0, // DEBUG级别
24+ LOG_LEVEL_INFO = 1, // INFO级别
25+ LOG_LEVEL_WARNING = 2, // WARNING级别
26+ LOG_LEVEL_ERROR = 3, // ERROR级别
27+ LOG_LEVEL_NONE = 4 // 关闭所有日志
28+} LogLevel;
29+ 
30+// CcuResult 返回码转换为 HcclResult 返回码
31+inline HcclResult ConvertCcuToHccl(CcuResult ccuResult)
32+{
33+ switch (ccuResult) {
34+ case CCU_SUCCESS:
35+ return HCCL_SUCCESS;
36+ case CCU_E_PARA:
37+ return HCCL_E_PARA;
38+ case CCU_E_PTR:
39+ return HCCL_E_PTR;
40+ case CCU_E_INTERNAL:
41+ return HCCL_E_INTERNAL;
42+ case CCU_E_NOT_SUPPORT:
43+ return HCCL_E_NOT_SUPPORT;
44+ case CCU_E_NOT_FOUND:
45+ return HCCL_E_NOT_FOUND;
46+ case CCU_E_UNAVAIL:
47+ return HCCL_E_UNAVAIL;
48+ default:
49+ return HCCL_E_INTERNAL;
50+ }
51+}
52+ 
53+#ifndef LIKELY
54+#define LIKELY(x) (static_cast<bool>(__builtin_expect(static_cast<bool>(x), 1)))
55+#define UNLIKELY(x) (static_cast<bool>(__builtin_expect(static_cast<bool>(x), 0)))
56+#endif
57+ 
58+#define HCCL_DEBUG(format, ...) \
59+ do { \
60+ if (LOG_LEVEL <= LOG_LEVEL_DEBUG) { \
61+ printf("[DEBUG][%s][%s:%d]" format "\n", __func__, __FILE__, __LINE__, ##__VA_ARGS__); \
62+ } \
63+ } while (0)
64+ 
65+#define HCCL_INFO(format, ...) \
66+ do { \
67+ if (LOG_LEVEL <= LOG_LEVEL_INFO) { \
68+ printf("[INFO][%s][%s:%d]" format "\n", __func__, __FILE__, __LINE__, ##__VA_ARGS__); \
69+ } \
70+ } while (0)
71+ 
72+#define HCCL_WARNING(format, ...) \
73+ do { \
74+ if (LOG_LEVEL <= LOG_LEVEL_WARNING) { \
75+ printf("[WARN][%s][%s:%d]" format "\n", __func__, __FILE__, __LINE__, ##__VA_ARGS__); \
76+ } \
77+ } while (0)
78+ 
79+#define HCCL_ERROR(format, ...) \
80+ do { \
81+ if (LOG_LEVEL <= LOG_LEVEL_ERROR) { \
82+ printf("[ERROR][%s][%s:%d]" format "\n", __func__, __FILE__, __LINE__, ##__VA_ARGS__); \
83+ } \
84+ } while (0)
85+ 
86+/* 检查指针, 若指针为NULL, 则记录日志, 并返回错误 */
87+#define CHK_PTR_NULL(ptr) \
88+ do { \
89+ if (UNLIKELY((ptr) == nullptr)) { \
90+ HCCL_ERROR("[%s] ptr [%s] is nullptr, return HCCL_E_PTR", __func__, #ptr); \
91+ return HCCL_E_PTR; \
92+ } \
93+ } while (0)
94+ 
95+/* 检查函数返回值, 记录指定日志, 并返回指定错误码 */
96+#define CHK_PRT_RET(result, exeLog, retCode) \
97+ do { \
98+ if (UNLIKELY(result)) { \
99+ exeLog; \
100+ return retCode; \
101+ } \
102+ } while (0)
103+ 
104+/* 检查函数返回值, 并返回指定错误码 */
105+#define CHK_RET(call) \
106+ do { \
107+ int32_t hcclRet = call; \
108+ if (UNLIKELY(hcclRet != HCCL_SUCCESS)) { \
109+ if (hcclRet == HCCL_E_AGAIN) { \
110+ HCCL_WARNING("[%s] call trace: hcclRet -> %d", __func__, hcclRet); \
111+ } else { \
112+ HCCL_ERROR("[%s] call trace: hcclRet -> %d", __func__, hcclRet); \
113+ } \
114+ return static_cast<HcclResult>(hcclRet); \
115+ } \
116+ } while (0)
117+ 
118+/* 检查函数返回值, 并返回指定错误码 */
119+#define CHK_RET_CCU(call) \
120+ do { \
121+ CcuResult ccuRet = call; \
122+ if (UNLIKELY(ccuRet != CCU_SUCCESS)) { \
123+ HCCL_ERROR("[%s] call trace: ccuRet -> %d", __func__, ccuRet); \
124+ return ConvertCcuToHccl(ccuRet); \
125+ } \
126+ } while (0)
127+ 
128+#define ACLCHECK(cmd) \
129+ do { \
130+ aclError ret = cmd; \
131+ if (UNLIKELY(ret != ACL_SUCCESS)) { \
132+ HCCL_ERROR("acl interface return err %s:%d, retcode: %d.\n", __FILE__, __LINE__, ret); \
133+ if (ret == ACL_ERROR_RT_MEMORY_ALLOCATION) { \
134+ HCCL_ERROR("memory allocation error, check whether the current memory space is sufficient.\n"); \
135+ } \
136+ return HCCL_E_RUNTIME; \
137+ } \
138+ } while (0)
139+ 
140+#endif // OPS_HCCL_LOG_H
@@ -0,0 +1,69 @@
1+# -----------------------------------------------------------------------------------------------------------
2+# Copyright (c) 2026 Huawei Technologies Co., Ltd.
3+# This program is free software, you can redistribute it and/or modify it under the terms and conditions of
4+# CANN Open Software License Agreement Version 2.0 (the "License").
5+# Please refer to the License for details. You may not use this file except in compliance with the License.
6+# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
7+# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
8+# See LICENSE in the root of the software repository for the full text of the License.
9+# -----------------------------------------------------------------------------------------------------------
10+ 
11+# ==================================================
12+# Host 侧动态链接库
13+# ==================================================
14+add_library(hccl SHARED)
15+ 
16+target_sources(hccl PRIVATE
17+ ${CMAKE_CURRENT_SOURCE_DIR}/reduce_scatter.cc
18+ ${CMAKE_CURRENT_SOURCE_DIR}/exec_op.cc
19+)
20+ 
21+target_include_directories(hccl PRIVATE
22+ ${CMAKE_CURRENT_SOURCE_DIR}/
23+ ${CMAKE_CURRENT_SOURCE_DIR}/../include
24+ # CANN Toolkit头文件
25+ ${ASCEND_CANN_PACKAGE_PATH}/include
26+ ${ASCEND_CANN_PACKAGE_PATH}/include/hccl
27+ ${ASCEND_CANN_PACKAGE_PATH}/include/hcomm
28+ ${ASCEND_CANN_PACKAGE_PATH}/include/hcomm/ccu
29+ ${ASCEND_CANN_PACKAGE_PATH}/pkg_inc
30+ ${ASCEND_CANN_PACKAGE_PATH}/pkg_inc/hccl
31+ ${ASCEND_CANN_PACKAGE_PATH}/pkg_inc/hcomm
32+ ${ASCEND_CANN_PACKAGE_PATH}/pkg_inc/hcomm/ccu
33+)
34+ 
35+target_compile_options(hccl PRIVATE
36+ -std=c++17
37+ -Werror
38+ -Wall
39+ -fno-common
40+ -fno-strict-aliasing
41+ -pipe
42+ -fstack-protector-all
43+ -U_FORTIFY_SOURCE
44+ $<$<CONFIG:Debug>:-g -O0>
45+ $<$<CONFIG:Release>:-O2>
46+)
47+ 
48+target_compile_definitions(hccl PRIVATE
49+ _GLIBCXX_USE_CXX11_ABI=0
50+)
51+ 
52+target_link_options(hccl PRIVATE
53+ -Wl,-z,relro
54+ -Wl,-z,now
55+ -Wl,-z,noexecstack
56+)
57+ 
58+target_link_directories(hccl PRIVATE
59+ ${ASCEND_CANN_PACKAGE_PATH}/lib64
60+)
61+ 
62+target_link_libraries(hccl PRIVATE
63+ hcomm
64+ acl_rt
65+)
66+ 
67+install(TARGETS hccl LIBRARY
68+ DESTINATION lib64
69+)
@@ -0,0 +1,282 @@
1+/**
2+ * Copyright (c) 2026 Huawei Technologies Co., Ltd.
3+ * This program is free software, you can redistribute it and/or modify it under the terms and conditions of
4+ * CANN Open Software License Agreement Version 2.0 (the "License").
5+ * Please refer to the License for details. You may not use this file except in compliance with the License.
6+ * THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
7+ * INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
8+ * See LICENSE in the root of the software repository for the full text of the License.
9+ */
10+ 
11+#include <limits>
12+#include <vector>
13+ 
14+#include <ccu/ccu_launch.h>
15+#include <ccu/ccu_res.h>
16+#include <hcomm/hcomm_primitives.h>
17+ 
18+#include "custom.h"
19+#include "exec_op.h"
20+ 
21+namespace ops_hccl {
22+namespace {
23+ constexpr uint64_t MAX_CHUNK_BYTES = 256ULL * 1024ULL * 1024ULL;
24+ constexpr uint64_t PARALLEL_COMBINE_MIN_BYTES = 256ULL * 1024ULL;
25+ constexpr uint64_t READ_PIPELINE_MIN_BYTES = 512ULL * 1024ULL;
26+ constexpr uint64_t READ_PIPELINE_ALIGNMENT = 4ULL * 1024ULL;
27+ constexpr uint64_t BUFFERED_REDUCE_MIN_BYTES = 256ULL * 1024ULL;
28+ constexpr uint64_t BUFFERED_REDUCE_SLICE_BYTES = 4ULL * 1024ULL;
29+ constexpr uint64_t BUFFERED_REDUCE_LOOP_COUNT = 16ULL;
30+ 
31+ struct BufferedReduceArgs {
32+ uint64_t enabled = 0;
33+ uint64_t addressOffset = 0;
34+ uint64_t loopIterations = 0;
35+ uint64_t parallelConfig = 0;
36+ uint64_t residualBytes = 0;
37+ };
38+ 
39+ constexpr uint64_t BitMask(uint16_t bitCount)
40+ {
41+ return (uint64_t{1} << bitCount) - 1;
42+ }
43+ 
44+ uint64_t PackParallelConfig(uint64_t repeatCount, uint64_t repeatLoopIndex, uint64_t totalLoopCount)
45+ {
46+ return ((repeatCount & BitMask(7)) << 55) | ((repeatLoopIndex & BitMask(7)) << 48)
47+ | ((totalLoopCount & BitMask(7)) << 41);
48+ }
49+ 
50+ BufferedReduceArgs CalcBufferedReduceArgs(uint64_t bytes)
51+ {
52+ BufferedReduceArgs args;
53+ if (bytes < BUFFERED_REDUCE_MIN_BYTES) {
54+ return args;
55+ }
56+ 
57+ constexpr uint64_t loopBytes = BUFFERED_REDUCE_SLICE_BYTES * BUFFERED_REDUCE_LOOP_COUNT;
58+ const uint64_t serialIterations = bytes / loopBytes;
59+ const uint64_t parallelSlices = (bytes % loopBytes) / BUFFERED_REDUCE_SLICE_BYTES;
60+ const uint64_t residualBytes = bytes % BUFFERED_REDUCE_SLICE_BYTES;
61+ 
62+ args.enabled = 1;
63+ args.addressOffset = serialIterations * loopBytes;
64+ args.loopIterations = serialIterations;
65+ if (parallelSlices == 0 && residualBytes == 0) {
66+ return args;
67+ }
68+ if (parallelSlices != 0 && residualBytes == 0) {
69+ args.parallelConfig = PackParallelConfig(parallelSlices - 1, 0, 1);
70+ args.residualBytes = BUFFERED_REDUCE_SLICE_BYTES;
71+ } else if (parallelSlices == 0) {
72+ args.parallelConfig = PackParallelConfig(0, 0, 1);
73+ args.residualBytes = residualBytes;
74+ } else {
75+ args.parallelConfig = PackParallelConfig(parallelSlices - 1, 1, 2);
76+ args.residualBytes = residualBytes;
77+ }
78+ return args;
79+ }
80+ 
81+ HcclResult ConvertCcuResult(CcuResult result)
82+ {
83+ switch (result) {
84+ case CCU_SUCCESS:
85+ return HCCL_SUCCESS;
86+ case CCU_E_PARA:
87+ return HCCL_E_PARA;
88+ case CCU_E_PTR:
89+ return HCCL_E_PTR;
90+ case CCU_E_NOT_SUPPORT:
91+ return HCCL_E_NOT_SUPPORT;
92+ case CCU_E_NOT_FOUND:
93+ return HCCL_E_NOT_FOUND;
94+ case CCU_E_UNAVAIL:
95+ return HCCL_E_UNAVAIL;
96+ default:
97+ return HCCL_E_INTERNAL;
98+ }
99+ }
100+ 
101+ HcclResult PreSyncThreads(ThreadHandle mainThread, ThreadHandle subThread)
102+ {
103+ HcclResult result
104+ = static_cast<HcclResult>(HcommThreadNotifyRecordOnThread(mainThread, subThread, 0));
105+ if (result != HCCL_SUCCESS) {
106+ return result;
107+ }
108+ return static_cast<HcclResult>(HcommThreadNotifyWaitOnThreadWithDefaultTimeout(subThread, 0));
109+ }
110+ 
111+ HcclResult PostSyncThreads(ThreadHandle mainThread, ThreadHandle subThread)
112+ {
113+ HcclResult result
114+ = static_cast<HcclResult>(HcommThreadNotifyWaitOnThreadWithDefaultTimeout(mainThread, 0));
115+ if (result != HCCL_SUCCESS) {
116+ return result;
117+ }
118+ return static_cast<HcclResult>(HcommThreadNotifyRecordOnThread(subThread, mainThread, 0));
119+ }
120+ 
121+ HcclResult LaunchKernel(ThreadHandle thread, CcuKernelHandle kernel, uint64_t inputAddress,
122+ uint64_t destinationAddress, uint64_t inputToken, uint64_t destinationToken, uint64_t scratchAddress,
123+ uint64_t scratchToken, uint64_t sourceBaseOffset, uint64_t chunkBytes, uint64_t pipelineFirstBytes,
124+ uint64_t pipelineSecondBytes, const BufferedReduceArgs &firstReduce,
125+ const BufferedReduceArgs &secondReduce)
126+ {
127+ const uint64_t taskArgs[] = {
128+ inputAddress, destinationAddress, inputToken, destinationToken, scratchAddress, scratchToken,
129+ sourceBaseOffset, chunkBytes, pipelineFirstBytes, pipelineSecondBytes,
130+ firstReduce.enabled, firstReduce.addressOffset, firstReduce.loopIterations,
131+ firstReduce.parallelConfig, firstReduce.residualBytes,
132+ secondReduce.enabled, secondReduce.addressOffset, secondReduce.loopIterations,
133+ secondReduce.parallelConfig, secondReduce.residualBytes};
134+ CcuResult launchResult = HcommCcuKernelLaunch(thread, kernel, taskArgs, sizeof(taskArgs) / sizeof(taskArgs[0]));
135+ if (launchResult != CCU_SUCCESS) {
136+ return ConvertCcuResult(launchResult);
137+ }
138+ return HCCL_SUCCESS;
139+ }
140+ 
141+ HcclResult LaunchCombineKernel(ThreadHandle thread, CcuKernelHandle kernel, uint64_t destinationAddress,
142+ uint64_t sourceAddress, uint64_t destinationToken, uint64_t sourceToken, uint64_t chunkBytes)
143+ {
144+ const uint64_t taskArgs[] = {
145+ destinationAddress, sourceAddress, destinationToken, sourceToken, chunkBytes};
146+ CcuResult launchResult = HcommCcuKernelLaunch(thread, kernel, taskArgs, sizeof(taskArgs) / sizeof(taskArgs[0]));
147+ if (launchResult != CCU_SUCCESS) {
148+ return ConvertCcuResult(launchResult);
149+ }
150+ return HCCL_SUCCESS;
151+ }
152+} // namespace
153+ 
154+HcclResult ExecOp(const OpParam &param)
155+{
156+ // 反序列化
157+ char *ctx = static_cast<char *>(param.resCtx);
158+ std::vector<char> seq(ctx, ctx + param.ctxSize);
159+ AlgResourceCtx resCtx;
160+ resCtx.DeSerialize(seq);
161+ 
162+ if (param.count == 0) {
163+ return HCCL_SUCCESS;
164+ }
165+ if (param.rankSize == 0 || param.rankSize > MAX_RANK_SIZE) {
166+ return HCCL_E_PARA;
167+ }
168+ 
169+ constexpr uint64_t dataTypeBytes = sizeof(float);
170+ if (param.count > std::numeric_limits<uint64_t>::max() / dataTypeBytes) {
171+ return HCCL_E_PARA;
172+ }
173+ const uint64_t recvBytes = param.count * dataTypeBytes;
174+ if (recvBytes > std::numeric_limits<uint64_t>::max() / param.rankSize) {
175+ return HCCL_E_PARA;
176+ }
177+ const uint64_t inputBytes = recvBytes * param.rankSize;
178+ 
179+ if (param.rankSize == 1) {
180+ return static_cast<HcclResult>(
181+ HcommLocalCopyOnThread(resCtx.threads[0], param.outputPtr, param.inputPtr, recvBytes));
182+ }
183+ if (resCtx.threads.empty() || resCtx.ccuKernels.empty()) {
184+ return HCCL_E_INTERNAL;
185+ }
186+ const bool useTwoDies = resCtx.threads.size() == 2 && resCtx.ccuKernels.size() == 4;
187+ if (!useTwoDies && (resCtx.threads.size() != 1 || resCtx.ccuKernels.size() != 1)) {
188+ return HCCL_E_INTERNAL;
189+ }
190+ 
191+ const uint64_t baseInputAddress = reinterpret_cast<uint64_t>(param.inputPtr);
192+ const uint64_t baseOutputAddress = reinterpret_cast<uint64_t>(param.outputPtr);
193+ uint64_t inputToken = 0;
194+ uint64_t outputToken = 0;
195+ CcuResult tokenResult = HcommCcuGetMemToken(baseInputAddress, inputBytes, &inputToken);
196+ if (tokenResult != CCU_SUCCESS) {
197+ return ConvertCcuResult(tokenResult);
198+ }
199+ tokenResult = HcommCcuGetMemToken(baseOutputAddress, recvBytes, &outputToken);
200+ if (tokenResult != CCU_SUCCESS) {
201+ return ConvertCcuResult(tokenResult);
202+ }
203+ 
204+ if (resCtx.localBuffer.addr == nullptr
205+ || resCtx.localBuffer.size < static_cast<uint64_t>(param.rankSize) * dataTypeBytes) {
206+ return HCCL_E_MEMORY;
207+ }
208+ 
209+ const uint64_t scratchSlotCount = param.rankSize;
210+ uint64_t maxChunkBytes = std::min(MAX_CHUNK_BYTES, resCtx.localBuffer.size / scratchSlotCount);
211+ maxChunkBytes -= maxChunkBytes % dataTypeBytes;
212+ if (maxChunkBytes == 0) {
213+ return HCCL_E_MEMORY;
214+ }
215+ const uint64_t scratchBytes = maxChunkBytes * scratchSlotCount;
216+ const uint64_t scratchAddress = reinterpret_cast<uint64_t>(resCtx.localBuffer.addr);
217+ uint64_t scratchToken = 0;
218+ tokenResult = HcommCcuGetMemToken(scratchAddress, scratchBytes, &scratchToken);
219+ if (tokenResult != CCU_SUCCESS) {
220+ return ConvertCcuResult(tokenResult);
221+ }
222+ 
223+ uint64_t processedBytes = 0;
224+ while (processedBytes < recvBytes) {
225+ const uint64_t chunkBytes = std::min(maxChunkBytes, recvBytes - processedBytes);
226+ const uint64_t outputAddress = baseOutputAddress + processedBytes;
227+ const uint64_t sourceBaseOffset = static_cast<uint64_t>(param.myRank) * recvBytes + processedBytes;
228+ uint64_t pipelineFirstBytes = 0;
229+ uint64_t pipelineSecondBytes = 0;
230+ if (chunkBytes >= READ_PIPELINE_MIN_BYTES) {
231+ pipelineFirstBytes = (chunkBytes / 2 / READ_PIPELINE_ALIGNMENT) * READ_PIPELINE_ALIGNMENT;
232+ if (pipelineFirstBytes == 0 || pipelineFirstBytes >= chunkBytes) {
233+ pipelineFirstBytes = 0;
234+ } else {
235+ pipelineSecondBytes = chunkBytes - pipelineFirstBytes;
236+ }
237+ }
238+ const uint64_t firstReduceBytes = pipelineFirstBytes == 0 ? chunkBytes : pipelineFirstBytes;
239+ const BufferedReduceArgs firstReduce = CalcBufferedReduceArgs(firstReduceBytes);
240+ const BufferedReduceArgs secondReduce = CalcBufferedReduceArgs(pipelineSecondBytes);
241+ 
242+ if (!useTwoDies) {
243+ CHK_RET(LaunchKernel(resCtx.threads[0], resCtx.ccuKernels[0], baseInputAddress, outputAddress, inputToken,
244+ outputToken, scratchAddress, scratchToken, sourceBaseOffset, chunkBytes, pipelineFirstBytes,
245+ pipelineSecondBytes, firstReduce, secondReduce));
246+ } else {
247+ const uint64_t partialAddress = scratchAddress;
248+ CHK_RET(PreSyncThreads(resCtx.threads[0], resCtx.threads[1]));
249+ CHK_RET(LaunchKernel(resCtx.threads[0], resCtx.ccuKernels[0], baseInputAddress, outputAddress, inputToken,
250+ outputToken, scratchAddress, scratchToken, sourceBaseOffset, chunkBytes, pipelineFirstBytes,
251+ pipelineSecondBytes, firstReduce, secondReduce));
252+ CHK_RET(LaunchKernel(resCtx.threads[1], resCtx.ccuKernels[1], baseInputAddress, partialAddress, inputToken,
253+ scratchToken, scratchAddress, scratchToken, sourceBaseOffset, chunkBytes, pipelineFirstBytes,
254+ pipelineSecondBytes, firstReduce, secondReduce));
255+ CHK_RET(PostSyncThreads(resCtx.threads[0], resCtx.threads[1]));
256+ 
257+ const uint64_t firstHalfBytes = pipelineFirstBytes != 0
258+ ? pipelineFirstBytes
259+ : (chunkBytes / dataTypeBytes / 2) * dataTypeBytes;
260+ const uint64_t secondHalfBytes = chunkBytes - firstHalfBytes;
261+ if (chunkBytes < PARALLEL_COMBINE_MIN_BYTES || firstHalfBytes == 0) {
262+ CHK_RET(LaunchCombineKernel(resCtx.threads[0], resCtx.ccuKernels[2], outputAddress, partialAddress,
263+ outputToken, scratchToken, chunkBytes));
264+ } else {
265+ const uint64_t secondPartialAddress = pipelineFirstBytes == 0
266+ ? partialAddress + firstHalfBytes
267+ : scratchAddress + static_cast<uint64_t>(param.rankSize) * pipelineFirstBytes;
268+ CHK_RET(PreSyncThreads(resCtx.threads[0], resCtx.threads[1]));
269+ CHK_RET(LaunchCombineKernel(resCtx.threads[0], resCtx.ccuKernels[2], outputAddress, partialAddress,
270+ outputToken, scratchToken, firstHalfBytes));
271+ CHK_RET(LaunchCombineKernel(resCtx.threads[1], resCtx.ccuKernels[3],
272+ outputAddress + firstHalfBytes, secondPartialAddress,
273+ outputToken, scratchToken, secondHalfBytes));
274+ CHK_RET(PostSyncThreads(resCtx.threads[0], resCtx.threads[1]));
275+ }
276+ }
277+ processedBytes += chunkBytes;
278+ }
279+ 
280+ return HCCL_SUCCESS;
281+}
282+} // namespace ops_hccl
@@ -0,0 +1,21 @@
1+/**
2+ * Copyright (c) 2026 Huawei Technologies Co., Ltd.
3+ * This program is free software, you can redistribute it and/or modify it under the terms and conditions of
4+ * CANN Open Software License Agreement Version 2.0 (the "License").
5+ * Please refer to the License for details. You may not use this file except in compliance with the License.
6+ * THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
7+ * INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
8+ * See LICENSE in the root of the software repository for the full text of the License.
9+ */
10+ 
11+#ifndef OPS_HCCL_CCU_EXEC_OP_H
12+#define OPS_HCCL_CCU_EXEC_OP_H
13+ 
14+#include <hccl/hcomm_primitives.h>
15+#include "common.h"
16+ 
17+namespace ops_hccl {
18+// 执行算法任务编排
19+HcclResult ExecOp(const OpParam &param);
20+} // namespace ops_hccl
21+#endif // OPS_HCCL_CCU_EXEC_OP_H
@@ -0,0 +1,343 @@
1+/**
2+ * Copyright (c) 2026 Huawei Technologies Co., Ltd.
3+ * This program is free software, you can redistribute it and/or modify it under the terms and conditions of
4+ * CANN Open Software License Agreement Version 2.0 (the "License").
5+ * Please refer to the License for details. You may not use this file except in compliance with the License.
6+ * THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
7+ * INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
8+ * See LICENSE in the root of the software repository for the full text of the License.
9+ */
10+ 
11+#include <array>
12+#include <algorithm>
13+#include <cstdio>
14+#include <memory>
15+#include <vector>
16+ 
17+#include <ccu/ccu_launch.h>
18+#include <hccl/hccl_ccu_res.h>
19+#include <hccl/hccl_diag.h>
20+#include <hccl/hccl_rank_graph.h>
21+#include <hccl/hccl_res_expt.h>
22+ 
23+#include "log.h"
24+#include "common.h"
25+#include "custom.h"
26+#include "hccl.h"
27+#include "exec_op.h"
28+#include "ccu_kernel.h"
29+ 
30+namespace {
31+constexpr uint32_t CHANNEL_NOTIFY_NUM = 3;
32+constexpr uint32_t MAX_DIE_NUM = 2;
33+ 
34+struct ChannelGroup {
35+ uint32_t dieId;
36+ std::vector<ChannelHandle> channels;
37+ std::vector<uint32_t> remoteRanks;
38+};
39+ 
40+// 用以持久化持有所有注册算子的参数指针,防止函数退出时被析构
41+static std::vector<std::shared_ptr<void>> g_persistentKernelArgs;
42+ 
43+HcclResult ConvertCcuResult(CcuResult result)
44+{
45+ switch (result) {
46+ case CCU_SUCCESS:
47+ return HCCL_SUCCESS;
48+ case CCU_E_PARA:
49+ return HCCL_E_PARA;
50+ case CCU_E_PTR:
51+ return HCCL_E_PTR;
52+ case CCU_E_NOT_SUPPORT:
53+ return HCCL_E_NOT_SUPPORT;
54+ case CCU_E_NOT_FOUND:
55+ return HCCL_E_NOT_FOUND;
56+ case CCU_E_UNAVAIL:
57+ return HCCL_E_UNAVAIL;
58+ default:
59+ return HCCL_E_INTERNAL;
60+ }
61+}
62+ 
63+HcclResult AcquirePeerChannels(
64+ HcclComm comm, CommEngine engine, uint32_t myRank, uint32_t rankSize, std::vector<ChannelGroup> &groups)
65+{
66+ uint32_t *networkLayers = nullptr;
67+ uint32_t networkLayerCount = 0;
68+ CHK_RET(HcclRankGraphGetLayers(comm, &networkLayers, &networkLayerCount));
69+ if (networkLayers == nullptr || networkLayerCount == 0) {
70+ HCCL_ERROR("[ReduceScatter] No topology layer is available for rank %u", myRank);
71+ return HCCL_E_NOT_FOUND;
72+ }
73+ const std::vector<uint32_t> networkLayerIds(networkLayers, networkLayers + networkLayerCount);
74+ 
75+ std::vector<HcclChannelDesc> channelDescs;
76+ std::vector<uint32_t> remoteRanks;
77+ std::vector<uint32_t> channelDieIds;
78+ channelDescs.reserve(rankSize - 1);
79+ remoteRanks.reserve(rankSize - 1);
80+ channelDieIds.reserve(rankSize - 1);
81+ 
82+ for (uint32_t remoteRank = 0; remoteRank < rankSize; ++remoteRank) {
83+ if (remoteRank == myRank) {
84+ continue;
85+ }
86+ 
87+ HcclChannelDesc desc;
88+ CHK_RET(HcclChannelDescInit(&desc, 1));
89+ bool protocolFound = false;
90+ 
91+ for (uint32_t layerIndex = 0; layerIndex < networkLayerCount && !protocolFound; ++layerIndex) {
92+ uint32_t linkCount = 0;
93+ CommLink *links = nullptr;
94+ HcclResult linkResult
95+ = HcclRankGraphGetLinks(comm, networkLayerIds[layerIndex], myRank, remoteRank, &links, &linkCount);
96+ if (linkResult != HCCL_SUCCESS || links == nullptr) {
97+ continue;
98+ }
99+ 
100+ for (uint32_t linkIndex = 0; linkIndex < linkCount; ++linkIndex) {
101+ const CommLink &link = links[linkIndex];
102+ if (link.linkAttr.linkProtocol != CommProtocol::COMM_PROTOCOL_UBC_CTP) {
103+ continue;
104+ }
105+ 
106+ desc.remoteRank = remoteRank;
107+ desc.notifyNum = CHANNEL_NOTIFY_NUM;
108+ desc.channelProtocol = link.linkAttr.linkProtocol;
109+ desc.localEndpoint.protocol = link.srcEndpointDesc.protocol;
110+ desc.localEndpoint.commAddr = link.srcEndpointDesc.commAddr;
111+ desc.localEndpoint.loc = link.srcEndpointDesc.loc;
112+ desc.remoteEndpoint.protocol = link.dstEndpointDesc.protocol;
113+ desc.remoteEndpoint.commAddr = link.dstEndpointDesc.commAddr;
114+ desc.remoteEndpoint.loc = link.dstEndpointDesc.loc;
115+ protocolFound = true;
116+ break;
117+ }
118+ }
119+ 
120+ if (!protocolFound) {
121+ HCCL_ERROR("[ReduceScatter] UBC_CTP link not found between rank %u and rank %u", myRank, remoteRank);
122+ return HCCL_E_NOT_FOUND;
123+ }
124+ 
125+ EndpointAttrDieId dieId = 0;
126+ uint32_t infoLength = sizeof(dieId);
127+ CHK_RET(HcclRankGraphGetEndpointInfo(
128+ comm, myRank, &desc.localEndpoint, ENDPOINT_ATTR_DIE_ID, infoLength, &dieId));
129+ if (dieId >= MAX_DIE_NUM) {
130+ HCCL_ERROR("[ReduceScatter] Invalid die id %u for channel from rank %u to rank %u", dieId, myRank,
131+ remoteRank);
132+ return HCCL_E_INTERNAL;
133+ }
134+ 
135+ channelDescs.push_back(desc);
136+ remoteRanks.push_back(remoteRank);
137+ channelDieIds.push_back(dieId);
138+ }
139+ 
140+ std::vector<ChannelHandle> channels(channelDescs.size());
141+ CHK_RET(HcclChannelAcquire(
142+ comm, engine, channelDescs.data(), static_cast<uint32_t>(channelDescs.size()), channels.data()));
143+ 
144+ std::array<ChannelGroup, MAX_DIE_NUM> groupsByDie;
145+ for (uint32_t dieId = 0; dieId < MAX_DIE_NUM; ++dieId) {
146+ groupsByDie[dieId].dieId = dieId;
147+ }
148+ for (uint32_t channelIndex = 0; channelIndex < channels.size(); ++channelIndex) {
149+ ChannelGroup &group = groupsByDie[channelDieIds[channelIndex]];
150+ group.channels.push_back(channels[channelIndex]);
151+ group.remoteRanks.push_back(remoteRanks[channelIndex]);
152+ }
153+ 
154+ groups.clear();
155+ for (uint32_t dieId = 0; dieId < MAX_DIE_NUM; ++dieId) {
156+ if (groupsByDie[dieId].channels.empty()) {
157+ continue;
158+ }
159+ groups.push_back(std::move(groupsByDie[dieId]));
160+ }
161+ if (groups.size() == MAX_DIE_NUM && groups[0].channels.size() > groups[1].channels.size()) {
162+ // The primary group also reduces the local rank. Put it on the die
163+ // with fewer peer channels to balance the number of sources.
164+ std::swap(groups[0], groups[1]);
165+ }
166+ 
167+ return HCCL_SUCCESS;
168+}
169+ 
170+HcclResult RegisterCcuKernels(
171+ HcclComm comm, const OpParam &param, const std::vector<ChannelGroup> &groups, AlgResourceCtx &resource)
172+{
173+ CcuInsHandle instructionHandle{0};
174+ uint32_t instructionCount = 0;
175+ CHK_RET(HcclCommQueryCcuIns(comm, &instructionHandle, &instructionCount));
176+ if (instructionCount != 1) {
177+ HCCL_ERROR("[ReduceScatter] Expected one CCU instruction instance, got %u", instructionCount);
178+ return HCCL_E_INTERNAL;
179+ }
180+ 
181+ CcuResult ccuResult = HcommCcuKernelRegisterStart(instructionHandle);
182+ if (ccuResult != CCU_SUCCESS) {
183+ return ConvertCcuResult(ccuResult);
184+ }
185+ 
186+ std::vector<std::shared_ptr<CcuReduceScatterKernelArg>> kernelArguments;
187+ kernelArguments.reserve(groups.size());
188+ const bool needsCombineKernel = groups.size() == MAX_DIE_NUM;
189+ resource.ccuKernels.resize(groups.size() + (needsCombineKernel ? groups.size() : 0));
190+ 
191+ for (uint32_t groupIndex = 0; groupIndex < groups.size(); ++groupIndex) {
192+ const ChannelGroup &group = groups[groupIndex];
193+ auto kernelArg = std::make_shared<CcuReduceScatterKernelArg>();
194+ kernelArg->rankSize = param.rankSize;
195+ kernelArg->rankId = param.myRank;
196+ kernelArg->includeLocal = groupIndex == 0;
197+ kernelArg->scratchSlotBase
198+ = needsCombineKernel && groupIndex == 0 ? static_cast<uint32_t>(groups[1].channels.size()) : 0;
199+ kernelArg->dataType = param.dataType;
200+ kernelArg->reduceOp = param.reduceType;
201+ kernelArg->channelCount = static_cast<uint32_t>(group.channels.size());
202+ for (uint32_t channelIndex = 0; channelIndex < group.channels.size(); ++channelIndex) {
203+ kernelArg->channels[channelIndex] = group.channels[channelIndex];
204+ kernelArg->remoteRanks[channelIndex] = group.remoteRanks[channelIndex];
205+ }
206+ 
207+ const void *kernelArgs[] = {kernelArg.get()};
208+ constexpr uint32_t registerDieId = 0; // Reserved by the registration API; channels select the actual die.
209+ ccuResult = HcommCcuKernelRegister(instructionHandle, registerDieId, "CcuReduceScatterKernel",
210+ reinterpret_cast<void *>(ops_hccl::CcuReduceScatterKernel), kernelArgs, 1,
211+ &resource.ccuKernels[groupIndex]);
212+ if (ccuResult != CCU_SUCCESS) {
213+ return ConvertCcuResult(ccuResult);
214+ }
215+ g_persistentKernelArgs.push_back(kernelArg);
216+ kernelArguments.push_back(std::move(kernelArg));
217+ }
218+ 
219+ if (needsCombineKernel) {
220+ for (uint32_t groupIndex = 0; groupIndex < groups.size(); ++groupIndex) {
221+ auto combineArg = std::make_shared<CcuCombineKernelArg>();
222+ combineArg->channelCount = 1;
223+ combineArg->channels[0] = groups[groupIndex].channels[0];
224+ const void *combineArgs[] = {combineArg.get()};
225+ ccuResult = HcommCcuKernelRegister(instructionHandle, 0, "CcuCombineKernel",
226+ reinterpret_cast<void *>(ops_hccl::CcuCombineKernel), combineArgs, 1,
227+ &resource.ccuKernels[groups.size() + groupIndex]);
228+ if (ccuResult != CCU_SUCCESS) {
229+ return ConvertCcuResult(ccuResult);
230+ }
231+ g_persistentKernelArgs.push_back(combineArg);
232+ }
233+ }
234+ 
235+ ccuResult = HcommCcuKernelRegisterEnd(instructionHandle);
236+ if (ccuResult != CCU_SUCCESS) {
237+ return ConvertCcuResult(ccuResult);
238+ }
239+ 
240+ return HCCL_SUCCESS;
241+}
242+} // namespace
243+ 
244+HcclResult HcclReduceScatter(void *sendBuf, void *recvBuf, uint64_t recvCount, HcclDataType dataType, HcclReduceOp op,
245+ HcclComm comm, aclrtStream stream)
246+{
247+ CHK_PTR_NULL(sendBuf);
248+ CHK_PTR_NULL(recvBuf);
249+ CHK_PTR_NULL(comm);
250+ CHK_PTR_NULL(stream);
251+ 
252+ if (dataType != HCCL_DATA_TYPE_FP32 || op != HCCL_REDUCE_SUM) {
253+ HCCL_ERROR("[ReduceScatter] Only float32 sum is supported");
254+ return HCCL_E_NOT_SUPPORT;
255+ }
256+ 
257+ // 构造算子参数
258+ OpParam param;
259+ int tagLength = std::snprintf(
260+ param.tag, sizeof(param.tag), "%s", "hccl_custom_reducescatter_ccu_buffered_reduce");
261+ if (tagLength <= 0 || static_cast<size_t>(tagLength) >= sizeof(param.tag)) {
262+ return HCCL_E_INTERNAL;
263+ }
264+ param.inputPtr = sendBuf;
265+ param.outputPtr = recvBuf;
266+ param.count = recvCount;
267+ param.dataType = dataType;
268+ param.opType = HcclCMDType::HCCL_CMD_REDUCE_SCATTER;
269+ param.reduceType = op;
270+ 
271+ // 注册算子信息
272+ HcclDfxOpInfo dfxInfo;
273+ char commName[COMM_INDENTIFIER_MAX_LENGTH];
274+ CHK_RET(HcclGetCommName(comm, commName));
275+ CHK_RET(HcclDfxRegOpInfoByCommId(commName, reinterpret_cast<void *>(&dfxInfo)));
276+ 
277+ // ==============================================
278+ // STEP 1: 解析拓扑信息
279+ // ==============================================
280+ CHK_RET(HcclGetRankId(comm, &param.myRank));
281+ CHK_RET(HcclGetRankSize(comm, &param.rankSize));
282+ 
283+ // ==============================================
284+ // STEP 2: 创建资源
285+ // ==============================================
286+ CommEngine ccuEngine = CommEngine::COMM_ENGINE_CCU;
287+ 
288+ // 将用户传入的 stream 转换为 CCU 通信引擎中的 thread
289+ CHK_RET(HcclThreadAcquireWithStream(comm, ccuEngine, stream, 1, &param.cpuThread));
290+ 
291+ void *ctx = nullptr;
292+ uint64_t size = 0;
293+ if (HcclEngineCtxGet(comm, param.tag, ccuEngine, &ctx, &size) == HCCL_SUCCESS) {
294+ // CCU 资源已经存在,复用资源
295+ HCCL_INFO("Engine context already exists");
296+ param.resCtx = ctx;
297+ param.ctxSize = size;
298+ } else {
299+ // Device 资源不存在,资源构建
300+ AlgResourceCtx resCtxHost{};
301+ resCtxHost.ccuThread = param.cpuThread;
302+ 
303+ // 从通信域获取 HCCL Buffer(Device上的内存,默认总大小400MB)
304+ void *cclBufferAddr;
305+ uint64_t cclBufferSize;
306+ CHK_RET(HcclGetHcclBuffer(comm, &cclBufferAddr, &cclBufferSize));
307+ resCtxHost.localBuffer = CommBuffer{cclBufferAddr, cclBufferSize};
308+ 
309+ if (param.rankSize > 1) {
310+ std::vector<ChannelGroup> channelGroups;
311+ CHK_RET(AcquirePeerChannels(comm, ccuEngine, param.myRank, param.rankSize, channelGroups));
312+ if (channelGroups.empty() || channelGroups.size() > MAX_DIE_NUM) {
313+ return HCCL_E_INTERNAL;
314+ }
315+ 
316+ resCtxHost.threads.resize(channelGroups.size());
317+ resCtxHost.threads[0] = param.cpuThread;
318+ if (channelGroups.size() > 1) {
319+ CHK_RET(HcclThreadAcquire(
320+ comm, ccuEngine, static_cast<uint32_t>(channelGroups.size() - 1), 1, &resCtxHost.threads[1]));
321+ }
322+ CHK_RET(RegisterCcuKernels(comm, param, channelGroups, resCtxHost));
323+ } else {
324+ resCtxHost.threads = {param.cpuThread};
325+ }
326+ 
327+ // ==============================================
328+ // STEP 2.3: 申请通信引擎上下文
329+ // ==============================================
330+ // 申请 CCU 通信引擎上下文,存放 AlgResourceCtx 信息
331+ std::vector<char> seq = resCtxHost.Serialize();
332+ uint64_t seqSize = seq.size();
333+ param.ctxSize = seqSize;
334+ CHK_RET(HcclEngineCtxCreate(comm, param.tag, ccuEngine, param.ctxSize, &param.resCtx));
335+ CHK_RET(HcclEngineCtxCopy(comm, ccuEngine, param.tag, seq.data(), seqSize, 0));
336+ }
337+ 
338+ // ==============================================
339+ // STEP 3: 下发 CCU Kernel
340+ // ==============================================
341+ CHK_RET(ops_hccl::ExecOp(param));
342+ return HCCL_SUCCESS;
343+}
@@ -0,0 +1,18 @@
1+ # -----------------------------------------------------------------------------------------------------------
2+# Copyright (c) 2026 Huawei Technologies Co., Ltd.
3+# This program is free software, you can redistribute it and/or modify it under the terms and conditions of
4+# CANN Open Software License Agreement Version 2.0 (the "License").
5+# Please refer to the License for details. You may not use this file except in compliance with the License.
6+# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
7+# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
8+# See LICENSE in the root of the software repository for the full text of the License.
9+# -----------------------------------------------------------------------------------------------------------
10+ 
11+if(TARGET hccl)
12+ target_sources(hccl PRIVATE
13+ ${CMAKE_CURRENT_SOURCE_DIR}/ccu_kernel.cc
14+ )
15+ target_include_directories(hccl PRIVATE
16+ ${CMAKE_CURRENT_SOURCE_DIR}
17+ )
18+endif()
@@ -0,0 +1,579 @@
1+/**
2+ * Copyright (c) 2026 Huawei Technologies Co., Ltd.
3+ * This program is free software, you can redistribute it and/or modify it under the terms and conditions of
4+ * CANN Open Software License Agreement Version 2.0 (the "License").
5+ * Please refer to the License for details. You may not use this file except in compliance with the License.
6+ * THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
7+ * INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
8+ * See LICENSE in the root of the software repository for the full text of the License.
9+ */
10+ 
11+#include <array>
12+#include <memory>
13+#include <vector>
14+ 
15+#include <hcomm/hcomm_primitives.h>
16+ 
17+#include "ccu_kernel.h"
18+ 
19+#ifndef CCU_CHK_RET
20+#define CCU_CHK_RET(call) \
21+ do { \
22+ CcuResult ccuResult = (call); \
23+ if (ccuResult != CCU_SUCCESS) { \
24+ return ccuResult; \
25+ } \
26+ } while (0)
27+#endif
28+ 
29+namespace ops_hccl {
30+namespace {
31+ constexpr uint16_t INPUT_ADDRESS_RESOURCE_ID = 0;
32+ constexpr uint16_t INPUT_TOKEN_RESOURCE_ID = 1;
33+ constexpr uint16_t POST_SYNC_ID = 2;
34+ constexpr uint16_t CHANNEL_EVENT_INDEX = 0;
35+ constexpr uint16_t TRANSFER_EVENT_MASK = 1;
36+ constexpr uint32_t BUFFERED_REDUCE_MAX_SOURCES = 8;
37+ constexpr uint32_t BUFFERED_REDUCE_LOOP_COUNT = 16;
38+ constexpr uint32_t BUFFERED_REDUCE_INTERLEAVE = 8;
39+ constexpr uint64_t BUFFERED_REDUCE_SLICE_BYTES = 4ULL * 1024ULL;
40+ 
41+ struct BufferedReduceGoSize {
42+ ccu::Variable enabled;
43+ ccu::Variable addressOffset;
44+ ccu::Variable loopIterations;
45+ ccu::Variable parallelConfig;
46+ ccu::Variable residualBytes;
47+ };
48+ 
49+ struct BufferedReduceResources {
50+ bool initialized = false;
51+ bool loopsCreated = false;
52+ ccu::Array<ccu::Event> completedEvents{0};
53+ ccu::Array<ccu::CcuBuffer> buffers{0};
54+ std::array<std::vector<ccu::LocalAddr>, 2> loopSources;
55+ ccu::LocalAddr loopDestination[2];
56+ ccu::Variable loopLength[2];
57+ ccu::Variable loopParameters[2];
58+ std::array<std::unique_ptr<ccu::Func>, 2> bodies;
59+ std::array<std::unique_ptr<ccu::Loop>, 2> loops;
60+ };
61+ 
62+ struct ReduceScatterContext {
63+ const CcuReduceScatterKernelArg *arg;
64+ std::vector<ccu::Variable> remoteInputAddresses;
65+ std::vector<ccu::Variable> remoteInputTokens;
66+ ccu::Variable localInputAddress;
67+ ccu::Variable localInputToken;
68+ ccu::Variable destinationAddress;
69+ ccu::Variable destinationToken;
70+ ccu::Variable scratchAddress;
71+ ccu::Variable scratchToken;
72+ ccu::Variable sourceBaseOffset;
73+ ccu::Variable chunkBytes;
74+ ccu::Variable pipelineFirstBytes;
75+ ccu::Variable pipelineSecondBytes;
76+ BufferedReduceGoSize firstReduce;
77+ BufferedReduceGoSize secondReduce;
78+ BufferedReduceResources bufferedReduce;
79+ ccu::Event event;
80+ ccu::Event pipelineEvent;
81+ };
82+ 
83+ constexpr uint64_t BitMask(uint16_t bitCount)
84+ {
85+ return (uint64_t{1} << bitCount) - 1;
86+ }
87+ 
88+ uint64_t PackLoopConfig(uint64_t addressOffset, uint64_t loopIterations)
89+ {
90+ return ((addressOffset & BitMask(32)) << 13) | (loopIterations & BitMask(13));
91+ }
92+ 
93+ uint64_t PackParallelConfig(uint64_t repeatCount, uint64_t repeatLoopIndex, uint64_t totalLoopCount)
94+ {
95+ return ((repeatCount & BitMask(7)) << 55) | ((repeatLoopIndex & BitMask(7)) << 48)
96+ | ((totalLoopCount & BitMask(7)) << 41);
97+ }
98+ 
99+ uint64_t PackOffsetConfig(uint64_t addressOffset, uint64_t bufferOffset, uint64_t eventOffset)
100+ {
101+ return ((addressOffset & BitMask(32)) << 21) | ((bufferOffset & BitMask(11)) << 10)
102+ | (eventOffset & BitMask(10));
103+ }
104+ 
105+ CcuResult InitResources(ReduceScatterContext &context)
106+ {
107+ const auto *arg = context.arg;
108+ context.remoteInputAddresses.resize(arg->channelCount);
109+ context.remoteInputTokens.resize(arg->channelCount);
110+ 
111+ for (uint32_t channel = 0; channel < arg->channelCount; ++channel) {
112+ context.remoteInputAddresses[channel]
113+ = ccu::GetResByChannel<ccu::Variable>(arg->channels[channel], INPUT_ADDRESS_RESOURCE_ID);
114+ context.remoteInputTokens[channel]
115+ = ccu::GetResByChannel<ccu::Variable>(arg->channels[channel], INPUT_TOKEN_RESOURCE_ID);
116+ }
117+ const uint32_t sourceCount = arg->channelCount + (arg->includeLocal ? 1U : 0U);
118+ if (sourceCount <= BUFFERED_REDUCE_MAX_SOURCES) {
119+ context.bufferedReduce.completedEvents = ccu::Array<ccu::Event>(BUFFERED_REDUCE_LOOP_COUNT);
120+ context.bufferedReduce.buffers = ccu::Array<ccu::CcuBuffer>(
121+ BUFFERED_REDUCE_LOOP_COUNT * BUFFERED_REDUCE_INTERLEAVE);
122+ context.bufferedReduce.initialized = true;
123+ }
124+ return CCU_SUCCESS;
125+ }
126+ 
127+ CcuResult LoadBufferedReduceArgs(BufferedReduceGoSize &goSize, uint32_t &argumentIndex)
128+ {
129+ CCU_CHK_RET(ccu::LoadArg(goSize.enabled, argumentIndex++));
130+ CCU_CHK_RET(ccu::LoadArg(goSize.addressOffset, argumentIndex++));
131+ CCU_CHK_RET(ccu::LoadArg(goSize.loopIterations, argumentIndex++));
132+ CCU_CHK_RET(ccu::LoadArg(goSize.parallelConfig, argumentIndex++));
133+ CCU_CHK_RET(ccu::LoadArg(goSize.residualBytes, argumentIndex++));
134+ return CCU_SUCCESS;
135+ }
136+ 
137+ CcuResult LoadArguments(ReduceScatterContext &context)
138+ {
139+ uint32_t argumentIndex = 0;
140+ CCU_CHK_RET(ccu::LoadArg(context.localInputAddress, argumentIndex++));
141+ CCU_CHK_RET(ccu::LoadArg(context.destinationAddress, argumentIndex++));
142+ CCU_CHK_RET(ccu::LoadArg(context.localInputToken, argumentIndex++));
143+ CCU_CHK_RET(ccu::LoadArg(context.destinationToken, argumentIndex++));
144+ CCU_CHK_RET(ccu::LoadArg(context.scratchAddress, argumentIndex++));
145+ CCU_CHK_RET(ccu::LoadArg(context.scratchToken, argumentIndex++));
146+ CCU_CHK_RET(ccu::LoadArg(context.sourceBaseOffset, argumentIndex++));
147+ CCU_CHK_RET(ccu::LoadArg(context.chunkBytes, argumentIndex++));
148+ CCU_CHK_RET(ccu::LoadArg(context.pipelineFirstBytes, argumentIndex++));
149+ CCU_CHK_RET(ccu::LoadArg(context.pipelineSecondBytes, argumentIndex++));
150+ CCU_CHK_RET(LoadBufferedReduceArgs(context.firstReduce, argumentIndex));
151+ CCU_CHK_RET(LoadBufferedReduceArgs(context.secondReduce, argumentIndex));
152+ return CCU_SUCCESS;
153+ }
154+ 
155+ CcuResult PreSync(ReduceScatterContext &context)
156+ {
157+ const auto *arg = context.arg;
158+ for (uint32_t channel = 0; channel < arg->channelCount; ++channel) {
159+ CCU_CHK_RET(ccu::WriteVariableWithNotify(arg->channels[channel], context.localInputAddress,
160+ INPUT_ADDRESS_RESOURCE_ID, CHANNEL_EVENT_INDEX, 1U << INPUT_ADDRESS_RESOURCE_ID));
161+ CCU_CHK_RET(ccu::WriteVariableWithNotify(arg->channels[channel], context.localInputToken,
162+ INPUT_TOKEN_RESOURCE_ID, CHANNEL_EVENT_INDEX, 1U << INPUT_TOKEN_RESOURCE_ID));
163+ }
164+ 
165+ constexpr uint32_t preSyncMask = (1U << INPUT_ADDRESS_RESOURCE_ID) | (1U << INPUT_TOKEN_RESOURCE_ID);
166+ for (uint32_t channel = 0; channel < arg->channelCount; ++channel) {
167+ CCU_CHK_RET(ccu::NotifyWait(arg->channels[channel], CHANNEL_EVENT_INDEX, preSyncMask));
168+ }
169+ return CCU_SUCCESS;
170+ }
171+ 
172+ CcuResult PostSync(ReduceScatterContext &context)
173+ {
174+ const auto *arg = context.arg;
175+ for (uint32_t channel = 0; channel < arg->channelCount; ++channel) {
176+ CCU_CHK_RET(ccu::NotifyRecord(arg->channels[channel], CHANNEL_EVENT_INDEX, 1U << POST_SYNC_ID));
177+ }
178+ for (uint32_t channel = 0; channel < arg->channelCount; ++channel) {
179+ CCU_CHK_RET(ccu::NotifyWait(arg->channels[channel], CHANNEL_EVENT_INDEX, 1U << POST_SYNC_ID));
180+ }
181+ return CCU_SUCCESS;
182+ }
183+ 
184+ CcuResult IssueSegmentReads(ReduceScatterContext &context, ccu::Variable scratchSegmentOffset,
185+ ccu::Variable sourceSegmentOffset, ccu::Variable segmentBytes, ccu::Event segmentEvent)
186+ {
187+ const auto *arg = context.arg;
188+ const uint32_t sourceCount = arg->channelCount + (arg->includeLocal ? 1U : 0U);
189+ ccu::LocalAddr scratchBase;
190+ scratchBase.addr = context.scratchAddress;
191+ scratchBase.addr += scratchSegmentOffset;
192+ for (uint32_t slot = 0; slot < arg->scratchSlotBase; ++slot) {
193+ scratchBase.addr += segmentBytes;
194+ }
195+ scratchBase.token = context.scratchToken;
196+ 
197+ uint32_t sourceIndex = 0;
198+ uint32_t readEventMask = 0;
199+ for (uint32_t sourceRank = 0; sourceRank < arg->rankSize; ++sourceRank) {
200+ ccu::LocalAddr scratchSlot;
201+ scratchSlot.addr = scratchBase.addr;
202+ for (uint32_t slot = 0; slot < sourceIndex; ++slot) {
203+ scratchSlot.addr += segmentBytes;
204+ }
205+ scratchSlot.token = context.scratchToken;
206+ 
207+ if (sourceRank == arg->rankId) {
208+ if (arg->includeLocal) {
209+ ccu::LocalAddr source;
210+ source.addr = context.localInputAddress;
211+ source.addr += context.sourceBaseOffset;
212+ source.addr += sourceSegmentOffset;
213+ source.token = context.localInputToken;
214+ 
215+ const uint32_t eventMask = 1U << sourceIndex;
216+ CCU_CHK_RET(ccu::LocalCopy(
217+ scratchSlot, source, segmentBytes, segmentEvent, eventMask));
218+ readEventMask |= eventMask;
219+ ++sourceIndex;
220+ }
221+ continue;
222+ }
223+ 
224+ for (uint32_t channel = 0; channel < arg->channelCount; ++channel) {
225+ if (arg->remoteRanks[channel] != sourceRank) {
226+ continue;
227+ }
228+ 
229+ ccu::RemoteAddr remoteSource;
230+ remoteSource.addr = context.remoteInputAddresses[channel];
231+ remoteSource.addr += context.sourceBaseOffset;
232+ remoteSource.addr += sourceSegmentOffset;
233+ remoteSource.token = context.remoteInputTokens[channel];
234+ 
235+ const uint32_t eventMask = 1U << sourceIndex;
236+ CCU_CHK_RET(ccu::Read(arg->channels[channel], scratchSlot, remoteSource, segmentBytes,
237+ segmentEvent, eventMask));
238+ readEventMask |= eventMask;
239+ ++sourceIndex;
240+ break;
241+ }
242+ }
243+ 
244+ if (sourceIndex != sourceCount || readEventMask == 0) {
245+ return CCU_E_PARA;
246+ }
247+ return CCU_SUCCESS;
248+ }
249+ 
250+ CcuResult CreateBufferedReduceLoops(ReduceScatterContext &context, uint32_t sourceCount)
251+ {
252+ auto &resources = context.bufferedReduce;
253+ if (!resources.initialized || sourceCount == 0 || sourceCount > BUFFERED_REDUCE_MAX_SOURCES) {
254+ return CCU_E_PARA;
255+ }
256+ if (resources.loopsCreated) {
257+ return CCU_SUCCESS;
258+ }
259+ 
260+ for (uint32_t index = 0; index < 2; ++index) {
261+ resources.loopSources[index].resize(sourceCount);
262+ const uint32_t bufferBase = index * BUFFERED_REDUCE_INTERLEAVE;
263+ ccu::Event loopEvent = resources.completedEvents[index];
264+ resources.bodies[index].reset(new ccu::Func(
265+ [&resources, index, bufferBase, loopEvent, sourceCount]() {
266+ for (uint32_t source = 0; source < sourceCount; ++source) {
267+ ccu::LocalCopy(resources.buffers[bufferBase + source],
268+ resources.loopSources[index][source], resources.loopLength[index], loopEvent,
269+ 1U << source);
270+ }
271+ ccu::EventWait(loopEvent, (1U << sourceCount) - 1U);
272+ 
273+ if (sourceCount > 1) {
274+ std::vector<ccu::CcuBuffer> reduceBuffers;
275+ reduceBuffers.reserve(sourceCount);
276+ for (uint32_t source = 0; source < sourceCount; ++source) {
277+ reduceBuffers.push_back(resources.buffers[bufferBase + source]);
278+ }
279+ ccu::LocalReduce(reduceBuffers.data(), sourceCount, HCCL_DATA_TYPE_FP32,
280+ HCCL_DATA_TYPE_FP32, HCCL_REDUCE_SUM, resources.loopLength[index], loopEvent,
281+ TRANSFER_EVENT_MASK);
282+ ccu::EventWait(loopEvent, TRANSFER_EVENT_MASK);
283+ }
284+ 
285+ ccu::LocalCopy(resources.loopDestination[index], resources.buffers[bufferBase],
286+ resources.loopLength[index], loopEvent, TRANSFER_EVENT_MASK);
287+ ccu::EventWait(loopEvent, TRANSFER_EVENT_MASK);
288+ }));
289+ resources.loops[index].reset(
290+ new ccu::Loop(resources.loopParameters[index], *resources.bodies[index]));
291+ }
292+ resources.loopsCreated = true;
293+ return CCU_SUCCESS;
294+ }
295+ 
296+ CcuResult RunBufferedReduce(ReduceScatterContext &context, ccu::LocalAddr destination,
297+ std::vector<ccu::LocalAddr> sources, BufferedReduceGoSize &goSize)
298+ {
299+ auto &resources = context.bufferedReduce;
300+ ccu::Variable loopConfig;
301+ ccu::Variable sliceBytes;
302+ ccu::Variable parallelConfig;
303+ ccu::Variable offsetConfig;
304+ 
305+ CCU_IF(goSize.loopIterations != 0)
306+ {
307+ loopConfig = PackLoopConfig(
308+ BUFFERED_REDUCE_SLICE_BYTES * BUFFERED_REDUCE_LOOP_COUNT, 0);
309+ loopConfig += goSize.loopIterations;
310+ sliceBytes = BUFFERED_REDUCE_SLICE_BYTES;
311+ for (uint32_t source = 0; source < sources.size(); ++source) {
312+ resources.loopSources[0][source].addr = sources[source].addr;
313+ resources.loopSources[0][source].token = sources[source].token;
314+ }
315+ resources.loopDestination[0].addr = destination.addr;
316+ resources.loopDestination[0].token = destination.token;
317+ resources.loopLength[0] = sliceBytes;
318+ parallelConfig = PackParallelConfig(BUFFERED_REDUCE_LOOP_COUNT - 1, 0, 1);
319+ offsetConfig = PackOffsetConfig(
320+ BUFFERED_REDUCE_SLICE_BYTES, BUFFERED_REDUCE_INTERLEAVE, 1);
321+ resources.loopParameters[0] = loopConfig;
322+ std::vector<ccu::Loop> loops{*resources.loops[0]};
323+ ccu::LoopGroup group(parallelConfig, offsetConfig, BUFFERED_REDUCE_LOOP_COUNT, loops);
324+ }
325+ 
326+ CCU_IF(goSize.parallelConfig != 0)
327+ {
328+ for (uint32_t source = 0; source < sources.size(); ++source) {
329+ sources[source].addr += goSize.addressOffset;
330+ }
331+ destination.addr += goSize.addressOffset;
332+ 
333+ for (uint32_t source = 0; source < sources.size(); ++source) {
334+ resources.loopSources[0][source].addr = sources[source].addr;
335+ resources.loopSources[0][source].token = sources[source].token;
336+ }
337+ resources.loopDestination[0].addr = destination.addr;
338+ resources.loopDestination[0].token = destination.token;
339+ resources.loopLength[0] = goSize.residualBytes;
340+ 
341+ for (uint32_t source = 0; source < sources.size(); ++source) {
342+ sources[source].addr += goSize.residualBytes;
343+ }
344+ destination.addr += goSize.residualBytes;
345+ sliceBytes = BUFFERED_REDUCE_SLICE_BYTES;
346+ for (uint32_t source = 0; source < sources.size(); ++source) {
347+ resources.loopSources[1][source].addr = sources[source].addr;
348+ resources.loopSources[1][source].token = sources[source].token;
349+ }
350+ resources.loopDestination[1].addr = destination.addr;
351+ resources.loopDestination[1].token = destination.token;
352+ resources.loopLength[1] = sliceBytes;
353+ 
354+ resources.loopParameters[0] = PackLoopConfig(0, 1);
355+ resources.loopParameters[1] = PackLoopConfig(0, 1);
356+ offsetConfig = PackOffsetConfig(
357+ BUFFERED_REDUCE_SLICE_BYTES, BUFFERED_REDUCE_INTERLEAVE, 1);
358+ std::vector<ccu::Loop> loops{*resources.loops[0], *resources.loops[1]};
359+ ccu::LoopGroup group(
360+ goSize.parallelConfig, offsetConfig, BUFFERED_REDUCE_LOOP_COUNT, loops);
361+ }
362+ return CCU_SUCCESS;
363+ }
364+ 
365+ CcuResult BufferedReduceSegment(ReduceScatterContext &context, ccu::Variable scratchSegmentOffset,
366+ ccu::Variable sourceSegmentOffset, ccu::Variable segmentBytes, ccu::Event segmentEvent,
367+ BufferedReduceGoSize &goSize)
368+ {
369+ const auto *arg = context.arg;
370+ const uint32_t sourceCount = arg->channelCount + (arg->includeLocal ? 1U : 0U);
371+ const uint32_t readEventMask = (1U << sourceCount) - 1U;
372+ CCU_CHK_RET(ccu::EventWait(segmentEvent, readEventMask));
373+ CCU_CHK_RET(CreateBufferedReduceLoops(context, sourceCount));
374+ 
375+ ccu::LocalAddr scratchBase;
376+ scratchBase.addr = context.scratchAddress;
377+ scratchBase.addr += scratchSegmentOffset;
378+ for (uint32_t slot = 0; slot < arg->scratchSlotBase; ++slot) {
379+ scratchBase.addr += segmentBytes;
380+ }
381+ scratchBase.token = context.scratchToken;
382+ 
383+ std::vector<ccu::LocalAddr> sources;
384+ sources.reserve(sourceCount);
385+ for (uint32_t source = 0; source < sourceCount; ++source) {
386+ ccu::LocalAddr sourceAddress;
387+ sourceAddress.addr = scratchBase.addr;
388+ for (uint32_t slot = 0; slot < source; ++slot) {
389+ sourceAddress.addr += segmentBytes;
390+ }
391+ sourceAddress.token = context.scratchToken;
392+ sources.push_back(sourceAddress);
393+ }
394+ 
395+ ccu::LocalAddr destination;
396+ if (arg->includeLocal) {
397+ destination.addr = context.destinationAddress;
398+ destination.addr += sourceSegmentOffset;
399+ destination.token = context.destinationToken;
400+ } else {
401+ destination.addr = scratchBase.addr;
402+ destination.token = scratchBase.token;
403+ }
404+ return RunBufferedReduce(context, destination, sources, goSize);
405+ }
406+ 
407+ CcuResult TreeReduceSegment(ReduceScatterContext &context, ccu::Variable scratchSegmentOffset,
408+ ccu::Variable sourceSegmentOffset, ccu::Variable segmentBytes, ccu::Event segmentEvent)
409+ {
410+ const auto *arg = context.arg;
411+ const uint32_t sourceCount = arg->channelCount + (arg->includeLocal ? 1U : 0U);
412+ const uint32_t readEventMask = (1U << sourceCount) - 1U;
413+ CCU_CHK_RET(ccu::EventWait(segmentEvent, readEventMask));
414+ 
415+ ccu::LocalAddr scratchBase;
416+ scratchBase.addr = context.scratchAddress;
417+ scratchBase.addr += scratchSegmentOffset;
418+ for (uint32_t slot = 0; slot < arg->scratchSlotBase; ++slot) {
419+ scratchBase.addr += segmentBytes;
420+ }
421+ scratchBase.token = context.scratchToken;
422+ 
423+ uint32_t remainingSources = sourceCount;
424+ while (remainingSources > 1) {
425+ const uint32_t reduceSources = remainingSources / 2;
426+ const uint32_t sourceSlot = remainingSources - reduceSources;
427+ 
428+ ccu::LocalAddr reduceSource;
429+ reduceSource.addr = scratchBase.addr;
430+ for (uint32_t slot = 0; slot < sourceSlot; ++slot) {
431+ reduceSource.addr += segmentBytes;
432+ }
433+ reduceSource.token = context.scratchToken;
434+ 
435+ ccu::Variable reduceBytes;
436+ reduceBytes = segmentBytes;
437+ for (uint32_t source = 1; source < reduceSources; ++source) {
438+ reduceBytes += segmentBytes;
439+ }
440+ 
441+ CCU_CHK_RET(ccu::LocalReduce(scratchBase, reduceSource, reduceBytes, arg->dataType, arg->reduceOp,
442+ segmentEvent, TRANSFER_EVENT_MASK));
443+ CCU_CHK_RET(ccu::EventWait(segmentEvent, TRANSFER_EVENT_MASK));
444+ remainingSources -= reduceSources;
445+ }
446+ 
447+ if (arg->includeLocal) {
448+ ccu::LocalAddr destination;
449+ destination.addr = context.destinationAddress;
450+ destination.addr += sourceSegmentOffset;
451+ destination.token = context.destinationToken;
452+ CCU_CHK_RET(
453+ ccu::LocalCopy(destination, scratchBase, segmentBytes, segmentEvent, TRANSFER_EVENT_MASK));
454+ CCU_CHK_RET(ccu::EventWait(segmentEvent, TRANSFER_EVENT_MASK));
455+ }
456+ return CCU_SUCCESS;
457+ }
458+ 
459+ CcuResult ReduceSegment(ReduceScatterContext &context, ccu::Variable scratchSegmentOffset,
460+ ccu::Variable sourceSegmentOffset, ccu::Variable segmentBytes, ccu::Event segmentEvent,
461+ BufferedReduceGoSize &goSize)
462+ {
463+ const uint32_t sourceCount
464+ = context.arg->channelCount + (context.arg->includeLocal ? 1U : 0U);
465+ if (sourceCount > BUFFERED_REDUCE_MAX_SOURCES) {
466+ return TreeReduceSegment(
467+ context, scratchSegmentOffset, sourceSegmentOffset, segmentBytes, segmentEvent);
468+ }
469+ 
470+ CCU_IF(goSize.enabled == 0)
471+ {
472+ CCU_CHK_RET(TreeReduceSegment(
473+ context, scratchSegmentOffset, sourceSegmentOffset, segmentBytes, segmentEvent));
474+ }
475+ CCU_IF(goSize.enabled != 0)
476+ {
477+ CCU_CHK_RET(BufferedReduceSegment(context, scratchSegmentOffset, sourceSegmentOffset,
478+ segmentBytes, segmentEvent, goSize));
479+ }
480+ return CCU_SUCCESS;
481+ }
482+ 
483+ CcuResult ExecuteReduceScatter(ReduceScatterContext &context)
484+ {
485+ ccu::Variable zero;
486+ zero = 0;
487+ 
488+ CCU_IF(context.pipelineFirstBytes == 0)
489+ {
490+ CCU_CHK_RET(IssueSegmentReads(context, zero, zero, context.chunkBytes, context.event));
491+ CCU_CHK_RET(ReduceSegment(
492+ context, zero, zero, context.chunkBytes, context.event, context.firstReduce));
493+ }
494+ 
495+ CCU_IF(context.pipelineFirstBytes != 0)
496+ {
497+ ccu::Variable secondScratchOffset;
498+ secondScratchOffset = context.pipelineFirstBytes;
499+ for (uint32_t rank = 1; rank < context.arg->rankSize; ++rank) {
500+ secondScratchOffset += context.pipelineFirstBytes;
501+ }
502+ 
503+ // Match the official mesh mem2mem schedule: issue the next read
504+ // batch before waiting for and reducing the current batch.
505+ CCU_CHK_RET(IssueSegmentReads(
506+ context, zero, zero, context.pipelineFirstBytes, context.event));
507+ CCU_CHK_RET(IssueSegmentReads(context, secondScratchOffset, context.pipelineFirstBytes,
508+ context.pipelineSecondBytes, context.pipelineEvent));
509+ CCU_CHK_RET(ReduceSegment(
510+ context, zero, zero, context.pipelineFirstBytes, context.event, context.firstReduce));
511+ CCU_CHK_RET(ReduceSegment(context, secondScratchOffset, context.pipelineFirstBytes,
512+ context.pipelineSecondBytes, context.pipelineEvent, context.secondReduce));
513+ }
514+ return CCU_SUCCESS;
515+ }
516+} // namespace
517+ 
518+CcuResult CcuReduceScatterKernel(CcuKernelArg kernelArgument)
519+{
520+ auto *kernelArg = static_cast<CcuReduceScatterKernelArg *>(kernelArgument);
521+ if (kernelArg == nullptr || kernelArg->rankSize == 0 || kernelArg->rankSize > MAX_RANK_SIZE
522+ || kernelArg->rankId >= kernelArg->rankSize || kernelArg->channelCount >= kernelArg->rankSize
523+ || (!kernelArg->includeLocal && kernelArg->channelCount == 0)
524+ || kernelArg->scratchSlotBase >= kernelArg->rankSize
525+ || kernelArg->scratchSlotBase + kernelArg->channelCount + (kernelArg->includeLocal ? 1U : 0U)
526+ > kernelArg->rankSize) {
527+ return CCU_E_PARA;
528+ }
529+ 
530+ ReduceScatterContext context;
531+ context.arg = kernelArg;
532+ 
533+ CCU_CHK_RET(InitResources(context));
534+ CCU_CHK_RET(LoadArguments(context));
535+ CCU_CHK_RET(PreSync(context));
536+ CCU_CHK_RET(ExecuteReduceScatter(context));
537+ CCU_CHK_RET(PostSync(context));
538+ 
539+ return CCU_SUCCESS;
540+}
541+ 
542+CcuResult CcuCombineKernel(CcuKernelArg kernelArgument)
543+{
544+ auto *kernelArg = static_cast<CcuCombineKernelArg *>(kernelArgument);
545+ if (kernelArg == nullptr || kernelArg->channelCount != 1) {
546+ return CCU_E_PARA;
547+ }
548+ 
549+ // Associate this local-only kernel with the same die as the primary
550+ // reduction kernel. The resource itself is not used for data transfer.
551+ (void)ccu::GetResByChannel<ccu::Variable>(kernelArg->channels[0], INPUT_ADDRESS_RESOURCE_ID);
552+ 
553+ ccu::Variable destinationAddress;
554+ ccu::Variable sourceAddress;
555+ ccu::Variable destinationToken;
556+ ccu::Variable sourceToken;
557+ ccu::Variable chunkBytes;
558+ uint32_t argumentIndex = 0;
559+ CCU_CHK_RET(ccu::LoadArg(destinationAddress, argumentIndex++));
560+ CCU_CHK_RET(ccu::LoadArg(sourceAddress, argumentIndex++));
561+ CCU_CHK_RET(ccu::LoadArg(destinationToken, argumentIndex++));
562+ CCU_CHK_RET(ccu::LoadArg(sourceToken, argumentIndex++));
563+ CCU_CHK_RET(ccu::LoadArg(chunkBytes, argumentIndex++));
564+ 
565+ ccu::LocalAddr destination;
566+ destination.addr = destinationAddress;
567+ destination.token = destinationToken;
568+ ccu::LocalAddr source;
569+ source.addr = sourceAddress;
570+ source.token = sourceToken;
571+ ccu::Event event;
572+ 
573+ CCU_CHK_RET(ccu::LocalReduce(destination, source, chunkBytes, HCCL_DATA_TYPE_FP32, HCCL_REDUCE_SUM, event,
574+ TRANSFER_EVENT_MASK));
575+ CCU_CHK_RET(ccu::EventWait(event, TRANSFER_EVENT_MASK));
576+ return CCU_SUCCESS;
577+}
578+ 
579+} // namespace ops_hccl
@@ -0,0 +1,27 @@
1+/**
2+ * Copyright (c) 2026 Huawei Technologies Co., Ltd.
3+ * This program is free software, you can redistribute it and/or modify it under the terms and conditions of
4+ * CANN Open Software License Agreement Version 2.0 (the "License").
5+ * Please refer to the License for details. You may not use this file except in compliance with the License.
6+ * THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
7+ * INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
8+ * See LICENSE in the root of the software repository for the full text of the License.
9+ */
10+ 
11+#ifndef OPS_HCCL_CCU_KERNEL_H
12+#define OPS_HCCL_CCU_KERNEL_H
13+ 
14+#include <ccu/ccu_types.h>
15+ 
16+#include "custom.h"
17+ 
18+namespace ccu = ::AscendC::ccu;
19+ 
20+namespace ops_hccl {
21+ 
22+// CCU Kernel 函数
23+CcuResult CcuReduceScatterKernel(CcuKernelArg arg);
24+CcuResult CcuCombineKernel(CcuKernelArg arg);
25+} // namespace ops_hccl
26+ 
27+#endif // OPS_HCCL_CCU_KERNEL_H
@@ -0,0 +1,106 @@
1+# -----------------------------------------------------------------------------------------------------------
2+# Copyright (c) 2026 Huawei Technologies Co., Ltd.
3+# This program is free software, you can redistribute it and/or modify it under the terms and conditions of
4+# CANN Open Software License Agreement Version 2.0 (the "License").
5+# Please refer to the License for details. You may not use this file except in compliance with the License.
6+# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
7+# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
8+# See LICENSE in the root of the software repository for the full text of the License.
9+# -----------------------------------------------------------------------------------------------------------
10+ 
11+Language: Cpp
12+Standard: c++17
13+TabWidth: 4
14+UseTab: Never
15+UseCRLF: false
16+AccessModifierOffset: -4
17+AlignConsecutiveAssignments: false
18+AlignConsecutiveDeclarations: false
19+AlignEscapedNewlines: DontAlign
20+AlignOperands: true
21+AlignTrailingComments: true
22+AllowAllArgumentsOnNextLine: true
23+AllowShortLambdasOnASingleLine: Empty
24+AllowAllParametersOfDeclarationOnNextLine: false
25+AllowShortBlocksOnASingleLine: false
26+AllowShortCaseLabelsOnASingleLine: false
27+AllowShortFunctionsOnASingleLine: false
28+AllowShortIfStatementsOnASingleLine: false
29+AllowShortLoopsOnASingleLine: false
30+AlwaysBreakAfterDefinitionReturnType: None
31+AlwaysBreakBeforeMultilineStrings: false
32+AlwaysBreakTemplateDeclarations: MultiLine
33+BinPackArguments: true
34+BinPackParameters: true
35+AlignAfterOpenBracket: DontAlign
36+ 
37+BraceWrapping:
38+ AfterCaseLabel: false
39+ AfterClass: false
40+ AfterControlStatement: Never
41+ AfterEnum: false
42+ AfterFunction: true
43+ AfterNamespace: false
44+ AfterStruct: false
45+ AfterUnion: false
46+ AfterExternBlock: false
47+ BeforeCatch: false
48+ BeforeElse: false
49+ IndentBraces: false
50+ SplitEmptyFunction: true
51+ SplitEmptyRecord: true
52+ SplitEmptyNamespace: true
53+ 
54+BreakBeforeBinaryOperators: All
55+BreakBeforeBraces: Custom
56+BreakBeforeTernaryOperators: true
57+BreakConstructorInitializersBeforeComma: false
58+BreakInheritanceList: AfterColon
59+ColumnLimit: 120
60+CommentPragmas: '^ IWYU pragma:'
61+PackConstructorInitializers: CurrentLine
62+ConstructorInitializerIndentWidth: 4
63+ContinuationIndentWidth: 4
64+Cpp11BracedListStyle: true
65+DerivePointerAlignment: false
66+DisableFormat: false
67+ExperimentalAutoDetectBinPacking: false
68+ForEachMacros: [ foreach ]
69+ 
70+IncludeCategories:
71+ - Regex: '^<'
72+ Priority: 3
73+ - Regex: '^"hccl'
74+ Priority: 2
75+ - Regex: '.*'
76+ Priority: 1
77+ 
78+IndentCaseLabels: true
79+IndentWidth: 4
80+IndentWrappedFunctionNames: false
81+KeepEmptyLinesAtTheStartOfBlocks: false
82+MacroBlockBegin: ''
83+MacroBlockEnd: ''
84+MaxEmptyLinesToKeep: 1
85+NamespaceIndentation: Inner
86+ 
87+PenaltyBreakBeforeFirstCallParameter: 19
88+PenaltyBreakComment: 300
89+PenaltyBreakFirstLessLess: 120
90+PenaltyBreakString: 1000
91+PenaltyExcessCharacter: 1000000
92+PenaltyReturnTypeOnItsOwnLine: 60
93+ 
94+PointerAlignment: Right
95+ReflowComments: true
96+SortIncludes: false
97+SpaceAfterCStyleCast: false
98+SpaceBeforeAssignmentOperators: true
99+SpaceBeforeParens: ControlStatements
100+SpaceInEmptyParentheses: false
101+SpacesBeforeTrailingComments: 1
102+SpacesInAngles: false
103+SpacesInContainerLiterals: true
104+SpacesInCStyleCastParentheses: false
105+SpacesInParentheses: false
106+SpacesInSquareBrackets: false
@@ -0,0 +1,11 @@
1+# -----------------------------------------------------------------------------------------------------------
2+# Copyright (c) 2026 Huawei Technologies Co., Ltd.
3+# This program is free software, you can redistribute it and/or modify it under the terms and conditions of
4+# CANN Open Software License Agreement Version 2.0 (the "License").
5+# Please refer to the License for details. You may not use this file except in compliance with the License.
6+# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
7+# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
8+# See LICENSE in the root of the software repository for the full text of the License.
9+# -----------------------------------------------------------------------------------------------------------
10+ 
11+* text=auto eol=lf
@@ -0,0 +1,35 @@
1+# -----------------------------------------------------------------------------------------------------------
2+# Copyright (c) 2026 Huawei Technologies Co., Ltd.
3+# This program is free software, you can redistribute it and/or modify it under the terms and conditions of
4+# CANN Open Software License Agreement Version 2.0 (the "License").
5+# Please refer to the License for details. You may not use this file except in compliance with the License.
6+# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
7+# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
8+# See LICENSE in the root of the software repository for the full text of the License.
9+# -----------------------------------------------------------------------------------------------------------
10+ 
11+# IDE
12+.vscode/
13+.idea/
14+.cache/
15+.DS_Store
16+*~
17+*.swp
18+*.swo
19+ 
20+# Build
21+build/
22+build_ut/
23+third_party/
24+__pycache__/
25+ 
26+# Agent
27+.omo/
28+.claude/
29+.opencode/
30+CLAUDE.md
31+AGENTS.md
32+ 
33+# Logs
34+logs/
35+*.log
@@ -0,0 +1,47 @@
1+# -----------------------------------------------------------------------------------------------------------
2+# Copyright (c) 2026 Huawei Technologies Co., Ltd.
3+# This program is free software, you can redistribute it and/or modify it under the terms and conditions of
4+# CANN Open Software License Agreement Version 2.0 (the "License").
5+# Please refer to the License for details. You may not use this file except in compliance with the License.
6+# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
7+# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
8+# See LICENSE in the root of the software repository for the full text of the License.
9+# -----------------------------------------------------------------------------------------------------------
10+ 
11+cmake_minimum_required(VERSION 3.16.0)
12+ project(hccl_reduce_scatter_aicpu)
13+ 
14+message(STATUS "CMAKE_BUILD_TYPE=${CMAKE_BUILD_TYPE}")
15+message(STATUS "ASCEND_CANN_PACKAGE_PATH=${ASCEND_CANN_PACKAGE_PATH}")
16+ 
17+set(HCC_TOOLCHAIN_DIR "${ASCEND_CANN_PACKAGE_PATH}/toolkit/toolchain/hcc")
18+set(HCC_CXX_COMPILER "${HCC_TOOLCHAIN_DIR}/bin/aarch64-target-linux-gnu-g++")
19+set(HCC_C_COMPILER "${HCC_TOOLCHAIN_DIR}/bin/aarch64-target-linux-gnu-gcc")
20+set(HCC_C_AR "${HCC_TOOLCHAIN_DIR}/bin/aarch64-target-linux-gnu-ar")
21+ 
22+# 编译 Host 侧链接库
23+add_subdirectory(op_host)
24+ 
25+# 编译 Device 侧链接库
26+include(ExternalProject)
27+ExternalProject_Add(hccl_device
28+ SOURCE_DIR ${CMAKE_SOURCE_DIR}/op_kernel_aicpu
29+ BINARY_DIR ${CMAKE_BINARY_DIR}/device_build
30+ CMAKE_ARGS
31+ -DTOOLCHAIN_DIR=${HCC_TOOLCHAIN_DIR}
32+ -DCMAKE_C_COMPILER=${HCC_C_COMPILER}
33+ -DCMAKE_CXX_COMPILER=${HCC_CXX_COMPILER}
34+ -DCMAKE_C_COMPILER_LAUNCHER=${CMAKE_C_COMPILER_LAUNCHER}
35+ -DCMAKE_CXX_COMPILER_LAUNCHER=${CMAKE_CXX_COMPILER_LAUNCHER}
36+ -DCMAKE_C_AR=${HCC_C_AR}
37+ -DCMAKE_BUILD_TYPE=${CMAKE_BUILD_TYPE}
38+ -DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}
39+ -DASCEND_CANN_PACKAGE_PATH=${ASCEND_CANN_PACKAGE_PATH}
40+ INSTALL_COMMAND ${CMAKE_COMMAND} --install ${CMAKE_BINARY_DIR}/device_build --prefix ${CMAKE_INSTALL_PREFIX}
41+ BUILD_ALWAYS TRUE
42+)
43+ 
44+# 安装
45+install(FILES include/hccl.h
46+ DESTINATION include
47+)
@@ -0,0 +1,201 @@
1+ Apache License
2+ Version 2.0, January 2004
3+ http://www.apache.org/licenses/
4+ 
5+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
6+ 
7+ 1. Definitions.
8+ 
9+ "License" shall mean the terms and conditions for use, reproduction,
10+ and distribution as defined by Sections 1 through 9 of this document.
11+ 
12+ "Licensor" shall mean the copyright owner or entity authorized by
13+ the copyright owner that is granting the License.
14+ 
15+ "Legal Entity" shall mean the union of the acting entity and all
16+ other entities that control, are controlled by, or are under common
17+ control with that entity. For the purposes of this definition,
18+ "control" means (i) the power, direct or indirect, to cause the
19+ direction or management of such entity, whether by contract or
20+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
21+ outstanding shares, or (iii) beneficial ownership of such entity.
22+ 
23+ "You" (or "Your") shall mean an individual or Legal Entity
24+ exercising permissions granted by this License.
25+ 
26+ "Source" form shall mean the preferred form for making modifications,
27+ including but not limited to software source code, documentation
28+ source, and configuration files.
29+ 
30+ "Object" form shall mean any form resulting from mechanical
31+ transformation or translation of a Source form, including but
32+ not limited to compiled object code, generated documentation,
33+ and conversions to other media types.
34+ 
35+ "Work" shall mean the work of authorship, whether in Source or
36+ Object form, made available under the License, as indicated by a
37+ copyright notice that is included in or attached to the work
38+ (an example is provided in the Appendix below).
39+ 
40+ "Derivative Works" shall mean any work, whether in Source or Object
41+ form, that is based on (or derived from) the Work and for which the
42+ editorial revisions, annotations, elaborations, or other modifications
43+ represent, as a whole, an original work of authorship. For the purposes
44+ of this License, Derivative Works shall not include works that remain
45+ separable from, or merely link (or bind by name) to the interfaces of,
46+ the Work and Derivative Works thereof.
47+ 
48+ "Contribution" shall mean any work of authorship, including
49+ the original version of the Work and any modifications or additions
50+ to that Work or Derivative Works thereof, that is intentionally
51+ submitted to Licensor for inclusion in the Work by the copyright owner
52+ or by an individual or Legal Entity authorized to submit on behalf of
53+ the copyright owner. For the purposes of this definition, "submitted"
54+ means any form of electronic, verbal, or written communication sent
55+ to the Licensor or its representatives, including but not limited to
56+ communication on electronic mailing lists, source code control systems,
57+ and issue tracking systems that are managed by, or on behalf of, the
58+ Licensor for the purpose of discussing and improving the Work, but
59+ excluding communication that is conspicuously marked or otherwise
60+ designated in writing by the copyright owner as "Not a Contribution."
61+ 
62+ "Contributor" shall mean Licensor and any individual or Legal Entity
63+ on behalf of whom a Contribution has been received by Licensor and
64+ subsequently incorporated within the Work.
65+ 
66+ 2. Grant of Copyright License. Subject to the terms and conditions of
67+ this License, each Contributor hereby grants to You a perpetual,
68+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
69+ copyright license to reproduce, prepare Derivative Works of,
70+ publicly display, publicly perform, sublicense, and distribute the
71+ Work and such Derivative Works in Source or Object form.
72+ 
73+ 3. Grant of Patent License. Subject to the terms and conditions of
74+ this License, each Contributor hereby grants to You a perpetual,
75+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
76+ (except as stated in this section) patent license to make, have made,
77+ use, offer to sell, sell, import, and otherwise transfer the Work,
78+ where such license applies only to those patent claims licensable
79+ by such Contributor that are necessarily infringed by their
80+ Contribution(s) alone or by combination of their Contribution(s)
81+ with the Work to which such Contribution(s) was submitted. If You
82+ institute patent litigation against any entity (including a
83+ cross-claim or counterclaim in a lawsuit) alleging that the Work
84+ or a Contribution incorporated within the Work constitutes direct
85+ or contributory patent infringement, then any patent licenses
86+ granted to You under this License for that Work shall terminate
87+ as of the date such litigation is filed.
88+ 
89+ 4. Redistribution. You may reproduce and distribute copies of the
90+ Work or Derivative Works thereof in any medium, with or without
91+ modifications, and in Source or Object form, provided that You
92+ meet the following conditions:
93+ 
94+ (a) You must give any other recipients of the Work or
95+ Derivative Works a copy of this License; and
96+ 
97+ (b) You must cause any modified files to carry prominent notices
98+ stating that You changed the files; and
99+ 
100+ (c) You must retain, in the Source form of any Derivative Works
101+ that You distribute, all copyright, patent, trademark, and
102+ attribution notices from the Source form of the Work,
103+ excluding those notices that do not pertain to any part of
104+ the Derivative Works; and
105+ 
106+ (d) If the Work includes a "NOTICE" text file as part of its
107+ distribution, then any Derivative Works that You distribute must
108+ include a readable copy of the attribution notices contained
109+ within such NOTICE file, excluding those notices that do not
110+ pertain to any part of the Derivative Works, in at least one
111+ of the following places: within a NOTICE text file distributed
112+ as part of the Derivative Works; within the Source form or
113+ documentation, if provided along with the Derivative Works; or,
114+ within a display generated by the Derivative Works, if and
115+ wherever such third-party notices normally appear. The contents
116+ of the NOTICE file are for informational purposes only and
117+ do not modify the License. You may add Your own attribution
118+ notices within Derivative Works that You distribute, alongside
119+ or as an addendum to the NOTICE text from the Work, provided
120+ that such additional attribution notices cannot be construed
121+ as modifying the License.
122+ 
123+ You may add Your own copyright statement to Your modifications and
124+ may provide additional or different license terms and conditions
125+ for use, reproduction, or distribution of Your modifications, or
126+ for any such Derivative Works as a whole, provided Your use,
127+ reproduction, and distribution of the Work otherwise complies with
128+ the conditions stated in this License.
129+ 
130+ 5. Submission of Contributions. Unless You explicitly state otherwise,
131+ any Contribution intentionally submitted for inclusion in the Work
132+ by You to the Licensor shall be under the terms and conditions of
133+ this License, without any additional terms or conditions.
134+ Notwithstanding the above, nothing herein shall supersede or modify
135+ the terms of any separate license agreement you may have executed
136+ with Licensor regarding such Contributions.
137+ 
138+ 6. Trademarks. This License does not grant permission to use the trade
139+ names, trademarks, service marks, or product names of the Licensor,
140+ except as required for reasonable and customary use in describing the
141+ origin of the Work and reproducing the content of the NOTICE file.
142+ 
143+ 7. Disclaimer of Warranty. Unless required by applicable law or
144+ agreed to in writing, Licensor provides the Work (and each
145+ Contributor provides its Contributions) on an "AS IS" BASIS,
146+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
147+ implied, including, without limitation, any warranties or conditions
148+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
149+ PARTICULAR PURPOSE. You are solely responsible for determining the
150+ appropriateness of using or redistributing the Work and assume any
151+ risks associated with Your exercise of permissions under this License.
152+ 
153+ 8. Limitation of Liability. In no event and under no legal theory,
154+ whether in tort (including negligence), contract, or otherwise,
155+ unless required by applicable law (such as deliberate and grossly
156+ negligent acts) or agreed to in writing, shall any Contributor be
157+ liable to You for damages, including any direct, indirect, special,
158+ incidental, or consequential damages of any character arising as a
159+ result of this License or out of the use or inability to use the
160+ Work (including but not limited to damages for loss of goodwill,
161+ work stoppage, computer failure or malfunction, or any and all
162+ other commercial damages or losses), even if such Contributor
163+ has been advised of the possibility of such damages.
164+ 
165+ 9. Accepting Warranty or Additional Liability. While redistributing
166+ the Work or Derivative Works thereof, You may choose to offer,
167+ and charge a fee for, acceptance of support, warranty, indemnity,
168+ or other liability obligations and/or rights consistent with this
169+ License. However, in accepting such obligations, You may act only
170+ on Your own behalf and on Your sole responsibility, not on behalf
171+ of any other Contributor, and only if You agree to indemnify,
172+ defend, and hold each Contributor harmless for any liability
173+ incurred by, or claims asserted against, such Contributor by reason
174+ of your accepting any such warranty or additional liability.
175+ 
176+ END OF TERMS AND CONDITIONS
177+ 
178+ APPENDIX: How to apply the Apache License to your work.
179+ 
180+ To apply the Apache License to your work, attach the following
181+ boilerplate notice, with the fields enclosed by brackets "{}"
182+ replaced with your own identifying information. (Don't include
183+ the brackets!) The text should be enclosed in the appropriate
184+ comment syntax for the file format. We also recommend that a
185+ file or class name and description of purpose be included on the
186+ same "printed page" as the copyright notice for easier
187+ identification within third-party archives.
188+ 
189+ Copyright {yyyy} {name of copyright owner}
190+ 
191+ Licensed under the Apache License, Version 2.0 (the "License");
192+ you may not use this file except in compliance with the License.
193+ You may obtain a copy of the License at
194+ 
195+ http://www.apache.org/licenses/LICENSE-2.0
196+ 
197+ Unless required by applicable law or agreed to in writing, software
198+ distributed under the License is distributed on an "AS IS" BASIS,
199+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
200+ See the License for the specific language governing permissions and
201+ limitations under the License.
@@ -0,0 +1,65 @@
1+# ReduceScatter 集合通信算子
2+ 
3+## 1. 项目介绍
4+ 
5+```
6+├── CMakeLists.txt # 顶层 CMake 配置
7+├── build.sh # 构建脚本
8+├── .clang-format # 代码风格配置
9+├── include/ # 头文件目录
10+│ ├── hccl.h # 集合通信算子头文件
11+│ ├── common.h # 通用数据结构定义
12+│ ├── custom.h # ★ 选手编写:自定义数据结构定义
13+│ ├── log.h # 日志宏定义
14+│ └── binary_stream.h # 序列化类定义
15+├── op_host/ # Host侧代码目录
16+│ ├── reduce_scatter.cc # ★ 选手编写:Host侧资源申请逻辑
17+│ └── launch_aicpu_kernel.cc # AICPU Kernel加载与下发逻辑
18+└── op_kernel_aicpu/ # Device侧代码目录
19+ ├── aicpu_kernel.cc # AICPU Kernel函数实现
20+ └── exec_op.cc # ★ 选手编写:通信算法编排逻辑
21+```
22+ 
23+> [!NOTE] 注意:
24+> 算子工程中已提前预制好固有逻辑,选手仅允许修改 `custom.h`、`reduce_scatter.cc`、`exec_op.cc` 共 3 个文件内容。
25+ 
26+## 2. 编译运行
27+ 
28+### 2.1 安装 CANN-Toolkit 包
29+ 
30+请单击[下载链接](https://ascend.devcloud.huaweicloud.com/artifactory/cann-run-mirror/software/legacy/20260701000328953/),根据产品型号和环境架构下载对应软件包。安装命令如下,更多指导参考《[CANN软件安装指南](https://www.hiascend.com/document/redirect/CannCommunityInstWizard)》。
31+ 
32+```bash
33+# 确保安装包具有可执行权限
34+chmod +x Ascend-cann-toolkit_9.1.0_linux-${arch}.run
35+# 安装命令
36+./Ascend-cann-toolkit_9.1.0_linux-${arch}.run --full --install-path=${install_path}
37+```
38+ 
39+### 2.2 环境变量配置
40+ 
41+按需选择合适的命令使环境变量生效。
42+ 
43+```bash
44+# 默认路径安装,以root用户为例(非root用户,将/usr/local替换为${HOME})
45+source /usr/local/Ascend/cann/set_env.sh
46+# 指定路径安装
47+# source ${install_path}/cann/set_env.sh
48+```
49+ 
50+### 2.3 编译算子工程
51+ 
52+```bash
53+bash build.sh
54+ 
55+# 编译 Debug 版本,便于断点调试
56+bash build.sh --debug
57+```
58+ 
59+## 3. 代码格式
60+ 
61+选手代码需符合 [.clang-format](.clang-format) 文件中的代码风格规范,可通过下列命令一键修改:
62+ 
63+```bash
64+bash build.sh --format
65+```
@@ -0,0 +1,10 @@
1+# 超能陆战队——初赛代码归档
2+ 
3+- 赛区:粤港澳赛区
4+- 学校:南方科技大学
5+- 队伍:超能陆战队
6+- 成员:Sami_Hui、Qu_、Alextory-Frank
7+- 赛题:ReduceScatter(AICPU + TS)
8+- 归档版本:`26746a8`
9+ 
10+本目录保留了初赛阶段实际开发的完整工程代码以及原始提交历史。实现针对两服务器拓扑完成了确定性 ReduceScatter,并对通信读取、切片和本地归约流程进行了优化。
@@ -0,0 +1,101 @@
1+# -----------------------------------------------------------------------------------------------------------
2+# Copyright (c) 2026 Huawei Technologies Co., Ltd.
3+# This program is free software, you can redistribute it and/or modify it under the terms and conditions of
4+# CANN Open Software License Agreement Version 2.0 (the "License").
5+# Please refer to the License for details. You may not use this file except in compliance with the License.
6+# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
7+# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
8+# See LICENSE in the root of the software repository for the full text of the License.
9+# -----------------------------------------------------------------------------------------------------------
10+ 
11+set -e
12+ 
13+PROJECT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
14+BUILD_DIR="${PROJECT_DIR}/build"
15+BUILD_TYPE="Release"
16+ENABLE_FORMAT="OFF"
17+ 
18+CPU_NUM="$(nproc)"
19+ASCEND_CANN_PACKAGE_PATH="/usr/local/Ascend/cann"
20+ 
21+usage() {
22+ cat <<EOF
23+Usage: $0 [OPTIONS]
24+ 
25+Options:
26+ --debug 编译 Debug 版本
27+ --format 格式化代码
28+ -h, --help 显示帮助信息
29+EOF
30+ exit 0
31+}
32+ 
33+parse_args() {
34+ local opts
35+ opts=$(getopt -o h -l debug,format,help -- "$@") || usage
36+ eval set -- "${opts}"
37+ 
38+ for arg; do
39+ case "${arg}" in
40+ --debug) BUILD_TYPE="Debug" ;;
41+ --format) ENABLE_FORMAT="ON" ;;
42+ -h|--help) usage ;;
43+ esac
44+ done
45+}
46+ 
47+parse_cann_path() {
48+ if [[ -z "${ASCEND_HOME_PATH}" ]]; then
49+ printf "ERROR: ASCEND_HOME_PATH is not set.\n" >&2
50+ printf "Please ensure CANN-Toolkit is properly installed and source environment variables by running:\n" >&2
51+ printf " source /path/to/Ascend/cann/set_env.sh\n" >&2
52+ exit 1
53+ fi
54+ 
55+ # 使用 local 避免污染全局(如需要全局可去掉 local)
56+ ASCEND_CANN_PACKAGE_PATH="${ASCEND_HOME_PATH}"
57+ return 0
58+}
59+ 
60+build() {
61+ # 创建构建目录
62+ cd "${PROJECT_DIR}"
63+ mkdir -p "${BUILD_DIR}"
64+ 
65+ # 配置
66+ cmake -S . -B "${BUILD_DIR}" \
67+ -DCMAKE_EXPORT_COMPILE_COMMANDS=ON \
68+ -DCMAKE_BUILD_TYPE="${BUILD_TYPE}" \
69+ -DCMAKE_INSTALL_PREFIX="${BUILD_DIR}" \
70+ -DASCEND_CANN_PACKAGE_PATH="${ASCEND_CANN_PACKAGE_PATH}"
71+ 
72+ # 编译
73+ cmake --build "${BUILD_DIR}" -j${CPU_NUM}
74+ 
75+ # 安装
76+ cmake --install "${BUILD_DIR}"
77+}
78+ 
79+format() {
80+ cd "${PROJECT_DIR}"
81+ find ./include -type f -regex '.*\.\(h\|hpp\|cpp\|cc\)$' -exec clang-format -i {} \;
82+ find ./op_host -type f -regex '.*\.\(h\|hpp\|cpp\|cc\)$' -exec clang-format -i {} \;
83+ find ./op_kernel_aicpu -type f -regex '.*\.\(h\|hpp\|cpp\|cc\)$' -exec clang-format -i {} \;
84+}
85+ 
86+main() {
87+ # 解析参数
88+ parse_args "$@"
89+ # 解析 CANN-Toolkit 路径
90+ parse_cann_path
91+ 
92+ if [[ "${ENABLE_FORMAT}" == "ON" ]]; then
93+ # 格式化代码
94+ format
95+ else
96+ # 编译
97+ build
98+ fi
99+}
100+ 
101+main "$@"