Skip to content

Commit b73daa7

Browse files
committed
CHS Integration for A3 Ultra
1 parent 85cf7e5 commit b73daa7

6 files changed

Lines changed: 187 additions & 2 deletions

File tree

examples/gke-a3-megagpu/gke-a3-megagpu-deployment.yaml

Lines changed: 2 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -46,5 +46,5 @@ vars:
4646
# can be inputted as <reservation-name>/reservationBlocks/<reservation-block-name>
4747
reservation: RESERVATION_NAME
4848

49-
enable_periodic_health_checks: false # Make this true to run CHS (healthchecks)
50-
health_check_schedule: "0 0 * * 0" # Run the health check at 12:00 AM (midnight) every Sunday
49+
enable_periodic_health_checks: # Make this true to run CHS (healthchecks)
50+
health_check_schedule: # Run the health check at 12:00 AM (midnight) every Sunday
Lines changed: 81 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,81 @@
1+
# Copyright 2024 "Google LLC"
2+
#
3+
# Licensed under the Apache License, Version 2.0 (the "License");
4+
# you may not use this file except in compliance with the License.
5+
# You may obtain a copy of the License at
6+
#
7+
# http://www.apache.org/licenses/LICENSE-2.0
8+
#
9+
# Unless required by applicable law or agreed to in writing, software
10+
# distributed under the License is distributed on an "AS IS" BASIS,
11+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12+
# See the License for the specific language governing permissions and
13+
# limitations under the License.
14+
15+
apiVersion: batch/v1
16+
kind: CronJob
17+
metadata:
18+
name: cluster-health-scanner-cronjob
19+
spec:
20+
schedule: "${cronjob_schedule}"
21+
concurrencyPolicy: Forbid
22+
successfulJobsHistoryLimit: 3
23+
failedJobsHistoryLimit: 1
24+
suspend: false
25+
jobTemplate:
26+
spec:
27+
template:
28+
spec:
29+
serviceAccountName: workload-identity-k8s-sa
30+
containers:
31+
- name: chs-runner
32+
image: python:3.11-slim-buster
33+
imagePullPolicy: Always
34+
command:
35+
- /bin/bash
36+
- -c
37+
- |
38+
set -ex
39+
set -x
40+
apt-get update && apt-get install -y git curl gnupg -y
41+
git clone https://github.com/GoogleCloudPlatform/cluster-health-scanner
42+
cd cluster-health-scanner
43+
apt-get install -y apt-transport-https ca-certificates
44+
curl https://packages.cloud.google.com/apt/doc/apt-key.gpg | gpg --dearmor -o /usr/share/keyrings/cloud.google.gpg
45+
echo "deb [signed-by=/usr/share/keyrings/cloud.google.gpg] https://packages.cloud.google.com/apt cloud-sdk main" | tee -a /etc/apt/sources.list.d/google-cloud-sdk.list
46+
apt-get update
47+
apt-get install -y google-cloud-cli kubectl
48+
apt-get install -y google-cloud-cli-gke-gcloud-auth-plugin
49+
curl https://raw.githubusercontent.com/helm/helm/main/scripts/get-helm-3 | bash
50+
pip3 install -r cli/requirements.txt
51+
gcloud container clusters get-credentials ${deployment_name} --region ${region} --project ${project_id}
52+
OUTPUT_DIR="/mnt/output"
53+
mkdir -p $OUTPUT_DIR
54+
TIMESTAMP="`date "+%Y-%m-%d %H:%M:%S"`"
55+
OUTPUT_FILENAME="${deployment_name}_healthscan_result_$TIMESTAMP.txt"
56+
FULL_OUTPUT_PATH="$OUTPUT_DIR/$OUTPUT_FILENAME"
57+
python3 cli/cluster_diag.py -o gke healthscan ${machine_type} -c gpu --run_only_on_available_nodes
58+
python3 cli/cluster_diag.py -o gke healthscan ${machine_type} -c nccl --run_only_on_available_nodes
59+
python3 cli/cluster_diag.py -o gke healthscan ${machine_type} -c straggler --run_only_on_available_nodes
60+
python3 cli/cluster_diag.py -o gke healthscan ${machine_type} -c neper --run_only_on_available_nodes
61+
python3 cli/cluster_diag.py -o gke healthscan ${machine_type} -c tinymax --run_only_on_available_nodes
62+
#python3 cli/cluster_diag.py -o gke healthscan ${machine_type} -c status --run_only_on_available_nodes > "$FULL_OUTPUT_PATH" 2>&1
63+
kubectl get nodes -o custom-columns="NODE:.metadata.name,NCCL_MARK:.metadata.labels.aiinfra/nccl-healthcheck-test,NCCL_BANDWIDTH:.metadata.labels.aiinfra/nccl-healthcheck-bandwidth,NCCL_RESULT:.metadata.labels.aiinfra/nccl-healthcheck-result,NCCL_RUNTIME:.metadata.labels.aiinfra/nccl-healthcheck-runtime-sec,TINYMAX_MARK:.metadata.labels.aiinfra/tinymax-healthcheck-test,TINYMAX_RESULT:.metadata.labels.aiinfra/tinymax-healthcheck-result,TINYMAX_RUNTIME:.metadata.labels.aiinfra/tinymax-healthcheck-runtime-sec,GPU_MARK:.metadata.labels.aiinfra/gpu-healthcheck-test,GPU_RESULT:.metadata.labels.aiinfra/gpu-healthcheck-result,GPU_RUNTIME:.metadata.labels.aiinfra/gpu-healthcheck-runtime-sec" > "$FULL_OUTPUT_PATH" 2>&1
64+
echo "Health scan outputs saved to $OUTPUT_DIR"
65+
echo "Final output file: $OUTPUT_FILENAME"
66+
volumeMounts:
67+
- name: ${gcs_bucket}
68+
mountPath: /mnt/output
69+
volumes:
70+
- name: ${gcs_bucket}
71+
persistentVolumeClaim:
72+
claimName: ${gcs_pvc}
73+
restartPolicy: Never
74+
tolerations:
75+
- key: "nvidia.com/gpu"
76+
operator: "Exists"
77+
effect: "NoSchedule"
78+
- key: "components.gke.io/gke-managed-components"
79+
operator: "Exists"
80+
effect: "NoSchedule"
81+
backoffLimit: 0
Lines changed: 59 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,59 @@
1+
apiVersion: v1
2+
kind: ServiceAccount
3+
metadata:
4+
name: workload-identity-k8s-sa
5+
namespace: default
6+
annotations:
7+
iam.gke.io/gcp-service-account: gke-wl-sa@${project_id}.iam.gserviceaccount.com
8+
---
9+
apiVersion: rbac.authorization.k8s.io/v1
10+
kind: ClusterRole
11+
metadata:
12+
name: cluster-health-scanner-job-role
13+
rules:
14+
- apiGroups: [""]
15+
resources:
16+
- "pods"
17+
- "pods/log"
18+
- "pods/exec"
19+
- "nodes"
20+
- "events"
21+
- "services"
22+
- "secrets"
23+
- "configmaps"
24+
- "serviceaccounts"
25+
verbs: ["list", "get", "create", "delete", "watch", "patch", "update"]
26+
27+
- apiGroups: ["apps"]
28+
resources:
29+
- "daemonsets"
30+
- "deployments"
31+
- "replicasets"
32+
verbs: ["list", "get", "create", "delete", "watch", "patch", "update"]
33+
34+
- apiGroups: ["batch"]
35+
resources:
36+
- "jobs"
37+
- "jobs/status"
38+
verbs: ["list", "get", "create", "delete", "watch", "patch", "update"]
39+
40+
- apiGroups: ["rbac.authorization.k8s.io"]
41+
resources:
42+
- "clusterrolebindings"
43+
- "clusterroles"
44+
- "roles"
45+
- "rolebindings"
46+
verbs: ["list", "get", "create", "delete", "watch", "patch", "update"]
47+
---
48+
apiVersion: rbac.authorization.k8s.io/v1
49+
kind: ClusterRoleBinding
50+
metadata:
51+
name: cluster-health-scanner-job-binding
52+
subjects:
53+
- kind: ServiceAccount
54+
name: workload-identity-k8s-sa
55+
namespace: default
56+
roleRef:
57+
kind: ClusterRole
58+
name: cluster-health-scanner-job-role
59+
apiGroup: rbac.authorization.k8s.io
Lines changed: 12 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,12 @@
1+
apiVersion: v1
2+
kind: PersistentVolumeClaim
3+
metadata:
4+
name: ${pvc_name}
5+
spec:
6+
accessModes:
7+
- ${access_mode}
8+
resources:
9+
requests:
10+
storage: ${capacity}
11+
storageClassName: ${storage_class_name}
12+

examples/gke-a3-ultragpu/gke-a3-ultragpu-deployment.yaml

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -36,3 +36,5 @@ vars:
3636
# disk for each node of the system node pool.
3737
a3ultra_node_pool_disk_size_gb: A3ULTRA_NODE_POOL_DISK_SIZE_GB # the size of
3838
# disk for each node.
39+
enable_periodic_health_checks: # Make this true to run CHS (healthchecks)
40+
health_check_schedule: # Run the health check at 12:00 AM (midnight) every Sunday

examples/gke-a3-ultragpu/gke-a3-ultragpu.yaml

Lines changed: 31 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -35,6 +35,15 @@ vars:
3535
accelerator_type: nvidia-h200-141gb
3636
version_prefix: "1.31."
3737

38+
enable_periodic_health_checks: false # Make this true to run CHS (healthchecks)
39+
health_check_schedule: "0 0 * * 0" # Run the health check at 12:00 AM (midnight) every Sunday
40+
41+
permissions_file_staged_path: $(ghpc_stage("./chs-permissions.yaml.tftpl"))
42+
chs_output_bucket_name: chs-result
43+
chs_pvc_claim_name: chs-output-pvc
44+
chs_cronjob_rendered_path: $(ghpc_stage("./chs-cronjob.yaml.tftpl"))
45+
chs_pvc_rendered_path: $(ghpc_stage("./chs-pvc.yaml.tftpl"))
46+
3847
deployment_groups:
3948
- group: primary
4049
modules:
@@ -116,6 +125,7 @@ deployment_groups:
116125
- stackdriver.resourceMetadata.writer
117126
- storage.objectAdmin
118127
- artifactregistry.reader
128+
- container.admin
119129

120130
- id: training_bucket
121131
source: community/modules/file-system/cloud-storage-bucket
@@ -214,6 +224,27 @@ deployment_groups:
214224
source: modules/management/kubectl-apply
215225
use: [a3-ultragpu-cluster]
216226
settings:
227+
apply_manifests:
228+
- source: $(vars.permissions_file_staged_path)
229+
template_vars:
230+
project_id: $(vars.project_id)
231+
- source: $(vars.chs_pvc_rendered_path)
232+
enable: $(vars.enable_periodic_health_checks)
233+
template_vars:
234+
pvc_name: $(vars.chs_pvc_claim_name)
235+
access_mode: ReadWriteOnce
236+
capacity: 1Gi
237+
storage_class_name: standard-rwo
238+
- source: $(vars.chs_cronjob_rendered_path)
239+
enable: $(vars.enable_periodic_health_checks)
240+
template_vars:
241+
project_id: $(vars.project_id)
242+
deployment_name: $(vars.deployment_name)
243+
region: $(vars.region)
244+
machine_type: a3-ultragpu-8g
245+
gcs_bucket: $(vars.chs_output_bucket_name)
246+
gcs_pvc: $(vars.chs_pvc_claim_name)
247+
cronjob_schedule: $(vars.health_check_schedule)
217248
kueue:
218249
install: true
219250
config_path: $(vars.kueue_configuration_path)

0 commit comments

Comments
 (0)