-
Notifications
You must be signed in to change notification settings - Fork 317
Expand file tree
/
Copy pathpfs-managed-lustre-slurm.yaml
More file actions
102 lines (92 loc) · 3.34 KB
/
Copy pathpfs-managed-lustre-slurm.yaml
File metadata and controls
102 lines (92 loc) · 3.34 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
# Copyright 2026 Google LLC
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
---
blueprint_name: managed-lustre-slurm
vars:
project_id: ## Set GCP Project ID Here ##
deployment_name: managed-lustre-slurm
region: us-central1
zone: us-central1-a
compute_node_machine_type: c2-standard-4
# Managed Lustre is only supported in specific regions and zones
# Please refer https://cloud.google.com/managed-lustre/docs/locations
# Managed-Lustre instance name. This should be unique for each deployment.
lustre_instance_id: lustre-instance
# The values of size_gib and per_unit_storage_throughput are co-related
# Please refer https://cloud.google.com/managed-lustre/docs/create-instance#performance-tiers
# Storage capacity of the lustre instance in GiB
lustre_size_gib: 36000
# Maximum throughput of the lustre instance in MBps per TiB
per_unit_storage_throughput: 500
kms_key: ""
slurm_image:
# Pinned to working version prior to July 10, 2026 due to Lustre client incompatibility on Rocky 9.8
name: slurm-gcp-6-12-hpc-rocky-linux-9-20260519
project: schedmd-slurm-public
# Documentation for each of the modules used below can be found at
# https://github.com/GoogleCloudPlatform/hpc-toolkit/blob/main/modules/README.md
deployment_groups:
- group: lustre
modules:
- id: network
source: modules/network/vpc
# Required for Managed Lustre instance
- id: private_service_access
source: modules/network/private-service-access
use: [network]
- id: lustre-gcp
source: modules/file-system/managed-lustre
use: [network, private_service_access]
settings:
name: $(vars.lustre_instance_id)
local_mount: /lustre
remote_mount: lustrefs
size_gib: $(vars.lustre_size_gib)
per_unit_storage_throughput: $(vars.per_unit_storage_throughput)
kms_key: $(vars.kms_key)
- group: slurm-cluster
modules:
- id: lustre-nodeset
source: community/modules/compute/schedmd-slurm-gcp-v6-nodeset
use: [network]
settings:
node_count_dynamic_max: 2
machine_type: $(vars.compute_node_machine_type)
allow_automatic_updates: false
instance_image: $(vars.slurm_image)
- id: lustre_partition
source: community/modules/compute/schedmd-slurm-gcp-v6-partition
use:
- lustre-nodeset
settings:
partition_name: lustre
is_default: true
- id: slurm_login
source: community/modules/scheduler/schedmd-slurm-gcp-v6-login
use: [network]
settings:
machine_type: n2-standard-4
enable_login_public_ips: false
instance_image: $(vars.slurm_image)
- id: slurm_controller
source: community/modules/scheduler/schedmd-slurm-gcp-v6-controller
use:
- network
- lustre_partition
- lustre-gcp
- slurm_login
settings:
machine_type: n2-standard-4
enable_controller_public_ips: false
instance_image: $(vars.slurm_image)