forked from NVIDIA/TransformerEngine
-
Notifications
You must be signed in to change notification settings - Fork 29
132 lines (115 loc) · 4.28 KB
/
Copy pathqa_cpp_distributed.yml
File metadata and controls
132 lines (115 loc) · 4.28 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
name: QA CPP Distributed
on:
push:
branches:
- main
pull_request:
branches:
- main
concurrency:
group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}-${{ github.actor }}
cancel-in-progress: true
jobs:
qa-l0-cpp_distributed:
runs-on: [ self-hosted, Linux, X64, nvida, gpu-8 ]
defaults:
run:
shell: bash
container:
image: harbor.baai.ac.cn/flagscale/cuda12.8.1-cudnn9.10.2-torch2.9.1-py3.10-transformerengine_fl:20260204
ports:
- 80:80
options: >-
--gpus all
--shm-size=500g
--privileged
--ipc=host
--ulimit memlock=-1
--ulimit stack=67108864
--ulimit nofile=65535:65535
--user root
--pull always
steps:
- name: Checkout Code
uses: actions/checkout@v6.0.1
with:
repository: ${{ github.event.pull_request.head.repo.full_name }}
ref: ${{ github.event.pull_request.head.ref }}
ssh-strict: true
ssh-user: git
persist-credentials: true
clean: true
sparse-checkout-cone-mode: true
fetch-tags: false
show-progress: true
lfs: false
submodules: recursive
set-safe-directory: true
- name: Pre-Check Environment
timeout-minutes: 5
run: |
set -euo pipefail
echo "===== System Info ====="
uname -a
python3 --version || (echo "❌ python3 not found"; exit 1)
pip3 --version || (echo "❌ pip3 not found"; exit 1)
echo "===== GPU Info ====="
if command -v nvidia-smi &>/dev/null; then
nvidia-smi --query-gpu=index,name,memory.total,utilization.gpu --format=csv
else
echo "⚠️ GPU check skipped. nvidia-smi is not available in the current environment."
exit 1
fi
nvcc --version || echo "⚠️ nvcc not found (non-fatal for python-only install)"
echo "===== Disk Space ====="
df -h
echo "===== Available memory ====="
free -h
- name: Install dependencies and build transformer_engine
# timeout-minutes: 30
env:
NVTE_FRAMEWORK: pytorch
TE_WITH_NCCL: 1
run: |
# =============== Install MPI ===============
apt update
apt install -y libopenmpi-dev openmpi-bin openmpi-common
apt install -y libmpich-dev mpich
# Verify the MPI header file
mpicxx -show | awk '{for(i=1;i<=NF;i++) if($i ~ /-I/) print substr($i,3)}'
# Verify whether the MPI C++ environment is ready
# 1. Verify whether the MPI C++ compiler (mpicxx) exists
mpicxx --version
# 2. Verify if the MPI library file exists
ls /usr/lib/x86_64-linux-gnu/libmpi_cxx.so
echo "=============== Install transformer_engine ==============="
pip install --no-build-isolation -vvv . --no-deps
cp -rf transformer_engine/plugins/ /usr/local/lib/python3.10/dist-packages/transformer_engine/plugins/
- name: Verify installation
shell: bash
run: |
python3 tests/pytorch/test_sanity_import.py
- name: GPU Usage Check / Verification
run: |
source .github/workflows/scripts/gpu_check.sh
wait_for_gpu
- name: L0 CPP Distributed
id: L0_cpp_distributed
# timeout-minutes: 10
env:
TE_PATH: .
XML_LOG_DIR: "/logs/cpp/distributed"
run: |
TE_LIB_PATH=$(pip3 show transformer-engine | grep -E "Location:|Editable project location:" | tail -n 1 | awk '{print $NF}')
TE_CPP_LIB_PATH="${TE_LIB_PATH}/transformer_engine"
export CMAKE_PREFIX_PATH="${TE_CPP_LIB_PATH}:${CMAKE_PREFIX_PATH}"
export LD_LIBRARY_PATH="${TE_CPP_LIB_PATH}:${LD_LIBRARY_PATH}"
bash ./qa/L1_cpp_distributed/test.sh | tee ${XML_LOG_DIR}/distributed-${{ github.run_id }}.log
- name: Upload Installation Logs
if: always() && steps.L0_cpp_distributed.outcome == 'failure'
uses: actions/upload-artifact@v4
with:
name: L0-cpp-logs-${{ github.run_id }}
path: /logs/cpp/distributed
retention-days: 7
if-no-files-found: warn