Skip to content

Commit 19342de

Browse files
authored
Merge pull request GoogleCloudPlatform#4323 from ishitachail/chs
CHS Integration for A4
2 parents 1576944 + c7f655b commit 19342de

5 files changed

Lines changed: 185 additions & 0 deletions

File tree

Lines changed: 81 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,81 @@
1+
# Copyright 2024 "Google LLC"
2+
#
3+
# Licensed under the Apache License, Version 2.0 (the "License");
4+
# you may not use this file except in compliance with the License.
5+
# You may obtain a copy of the License at
6+
#
7+
# http://www.apache.org/licenses/LICENSE-2.0
8+
#
9+
# Unless required by applicable law or agreed to in writing, software
10+
# distributed under the License is distributed on an "AS IS" BASIS,
11+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12+
# See the License for the specific language governing permissions and
13+
# limitations under the License.
14+
15+
apiVersion: batch/v1
16+
kind: CronJob
17+
metadata:
18+
name: cluster-health-scanner-cronjob
19+
spec:
20+
schedule: "${cronjob_schedule}"
21+
concurrencyPolicy: Forbid
22+
successfulJobsHistoryLimit: 3
23+
failedJobsHistoryLimit: 1
24+
suspend: false
25+
jobTemplate:
26+
spec:
27+
template:
28+
spec:
29+
serviceAccountName: workload-identity-k8s-sa
30+
containers:
31+
- name: chs-runner
32+
image: python:3.11-slim-buster
33+
imagePullPolicy: Always
34+
command:
35+
- /bin/bash
36+
- -c
37+
- |
38+
set -ex
39+
set -x
40+
apt-get update && apt-get install -y git curl gnupg -y
41+
git clone https://github.com/GoogleCloudPlatform/cluster-health-scanner
42+
cd cluster-health-scanner
43+
apt-get install -y apt-transport-https ca-certificates
44+
curl https://packages.cloud.google.com/apt/doc/apt-key.gpg | gpg --dearmor -o /usr/share/keyrings/cloud.google.gpg
45+
echo "deb [signed-by=/usr/share/keyrings/cloud.google.gpg] https://packages.cloud.google.com/apt cloud-sdk main" | tee -a /etc/apt/sources.list.d/google-cloud-sdk.list
46+
apt-get update
47+
apt-get install -y google-cloud-cli kubectl
48+
apt-get install -y google-cloud-cli-gke-gcloud-auth-plugin
49+
curl https://raw.githubusercontent.com/helm/helm/main/scripts/get-helm-3 | bash
50+
pip3 install -r cli/requirements.txt
51+
gcloud container clusters get-credentials ${deployment_name} --region ${region} --project ${project_id}
52+
OUTPUT_DIR="/mnt/output"
53+
mkdir -p $OUTPUT_DIR
54+
TIMESTAMP="`date "+%Y-%m-%d %H:%M:%S"`"
55+
OUTPUT_FILENAME="${deployment_name}_healthscan_result_$TIMESTAMP.txt"
56+
FULL_OUTPUT_PATH="$OUTPUT_DIR/$OUTPUT_FILENAME"
57+
python3 cli/cluster_diag.py -o gke healthscan ${machine_type} -c gpu --run_only_on_available_nodes
58+
python3 cli/cluster_diag.py -o gke healthscan ${machine_type} -c nccl --run_only_on_available_nodes
59+
python3 cli/cluster_diag.py -o gke healthscan ${machine_type} -c straggler --run_only_on_available_nodes
60+
python3 cli/cluster_diag.py -o gke healthscan ${machine_type} -c neper --run_only_on_available_nodes
61+
python3 cli/cluster_diag.py -o gke healthscan ${machine_type} -c tinymax --run_only_on_available_nodes
62+
#python3 cli/cluster_diag.py -o gke healthscan ${machine_type} -c status --run_only_on_available_nodes > "$FULL_OUTPUT_PATH" 2>&1
63+
kubectl get nodes -o custom-columns="NODE:.metadata.name,NCCL_MARK:.metadata.labels.aiinfra/nccl-healthcheck-test,NCCL_BANDWIDTH:.metadata.labels.aiinfra/nccl-healthcheck-bandwidth,NCCL_RESULT:.metadata.labels.aiinfra/nccl-healthcheck-result,NCCL_RUNTIME:.metadata.labels.aiinfra/nccl-healthcheck-runtime-sec,TINYMAX_MARK:.metadata.labels.aiinfra/tinymax-healthcheck-test,TINYMAX_RESULT:.metadata.labels.aiinfra/tinymax-healthcheck-result,TINYMAX_RUNTIME:.metadata.labels.aiinfra/tinymax-healthcheck-runtime-sec,GPU_MARK:.metadata.labels.aiinfra/gpu-healthcheck-test,GPU_RESULT:.metadata.labels.aiinfra/gpu-healthcheck-result,GPU_RUNTIME:.metadata.labels.aiinfra/gpu-healthcheck-runtime-sec" > "$FULL_OUTPUT_PATH" 2>&1
64+
echo "Health scan outputs saved to $OUTPUT_DIR"
65+
echo "Final output file: $OUTPUT_FILENAME"
66+
volumeMounts:
67+
- name: ${gcs_bucket}
68+
mountPath: /mnt/output
69+
volumes:
70+
- name: ${gcs_bucket}
71+
persistentVolumeClaim:
72+
claimName: ${gcs_pvc}
73+
restartPolicy: Never
74+
tolerations:
75+
- key: "nvidia.com/gpu"
76+
operator: "Exists"
77+
effect: "NoSchedule"
78+
- key: "components.gke.io/gke-managed-components"
79+
operator: "Exists"
80+
effect: "NoSchedule"
81+
backoffLimit: 0
Lines changed: 59 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,59 @@
1+
apiVersion: v1
2+
kind: ServiceAccount
3+
metadata:
4+
name: workload-identity-k8s-sa
5+
namespace: default
6+
annotations:
7+
iam.gke.io/gcp-service-account: gke-wl-sa@${project_id}.iam.gserviceaccount.com
8+
---
9+
apiVersion: rbac.authorization.k8s.io/v1
10+
kind: ClusterRole
11+
metadata:
12+
name: cluster-health-scanner-job-role
13+
rules:
14+
- apiGroups: [""]
15+
resources:
16+
- "pods"
17+
- "pods/log"
18+
- "pods/exec"
19+
- "nodes"
20+
- "events"
21+
- "services"
22+
- "secrets"
23+
- "configmaps"
24+
- "serviceaccounts"
25+
verbs: ["list", "get", "create", "delete", "watch", "patch", "update"]
26+
27+
- apiGroups: ["apps"]
28+
resources:
29+
- "daemonsets"
30+
- "deployments"
31+
- "replicasets"
32+
verbs: ["list", "get", "create", "delete", "watch", "patch", "update"]
33+
34+
- apiGroups: ["batch"]
35+
resources:
36+
- "jobs"
37+
- "jobs/status"
38+
verbs: ["list", "get", "create", "delete", "watch", "patch", "update"]
39+
40+
- apiGroups: ["rbac.authorization.k8s.io"]
41+
resources:
42+
- "clusterrolebindings"
43+
- "clusterroles"
44+
- "roles"
45+
- "rolebindings"
46+
verbs: ["list", "get", "create", "delete", "watch", "patch", "update"]
47+
---
48+
apiVersion: rbac.authorization.k8s.io/v1
49+
kind: ClusterRoleBinding
50+
metadata:
51+
name: cluster-health-scanner-job-binding
52+
subjects:
53+
- kind: ServiceAccount
54+
name: workload-identity-k8s-sa
55+
namespace: default
56+
roleRef:
57+
kind: ClusterRole
58+
name: cluster-health-scanner-job-role
59+
apiGroup: rbac.authorization.k8s.io

examples/gke-a4/chs-pvc.yaml.tftpl

Lines changed: 12 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,12 @@
1+
apiVersion: v1
2+
kind: PersistentVolumeClaim
3+
metadata:
4+
name: ${pvc_name}
5+
spec:
6+
accessModes:
7+
- ${access_mode}
8+
resources:
9+
requests:
10+
storage: ${capacity}
11+
storageClassName: ${storage_class_name}
12+

examples/gke-a4/gke-a4-deployment.yaml

Lines changed: 3 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -51,3 +51,6 @@ vars:
5151

5252
# The disk size of a4 node pool for this deployment.
5353
a4_node_pool_disk_size_gb:
54+
55+
enable_periodic_health_checks: # Make this true to run CHS (healthchecks)
56+
health_check_schedule: # Run the health check at 12:00 AM (midnight) every Sunday

examples/gke-a4/gke-a4.yaml

Lines changed: 30 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -50,6 +50,14 @@ vars:
5050
accelerator_type: nvidia-b200
5151
version_prefix: "1.32."
5252

53+
enable_periodic_health_checks: false # Make this true to run CHS (healthchecks)
54+
health_check_schedule: "0 0 * * 0" # Run the health check at 12:00 AM (midnight) every Sunday
55+
56+
permissions_file_staged_path: $(ghpc_stage("./chs-permissions.yaml.tftpl"))
57+
chs_output_bucket_name: chs-result
58+
chs_pvc_claim_name: chs-output-pvc
59+
chs_cronjob_rendered_path: $(ghpc_stage("./chs-cronjob.yaml.tftpl"))
60+
chs_pvc_rendered_path: $(ghpc_stage("./chs-pvc.yaml.tftpl"))
5361

5462
deployment_groups:
5563
- group: primary
@@ -132,6 +140,7 @@ deployment_groups:
132140
- stackdriver.resourceMetadata.writer
133141
- storage.objectAdmin
134142
- artifactregistry.reader
143+
- container.admin
135144

136145
- id: training_bucket
137146
source: community/modules/file-system/cloud-storage-bucket
@@ -231,6 +240,27 @@ deployment_groups:
231240
source: modules/management/kubectl-apply
232241
use: [a4-cluster]
233242
settings:
243+
apply_manifests:
244+
- source: $(vars.permissions_file_staged_path)
245+
template_vars:
246+
project_id: $(vars.project_id)
247+
- source: $(vars.chs_pvc_rendered_path)
248+
enable: $(vars.enable_periodic_health_checks)
249+
template_vars:
250+
pvc_name: $(vars.chs_pvc_claim_name)
251+
access_mode: ReadWriteOnce
252+
capacity: 1Gi
253+
storage_class_name: standard-rwo
254+
- source: $(vars.chs_cronjob_rendered_path)
255+
enable: $(vars.enable_periodic_health_checks)
256+
template_vars:
257+
project_id: $(vars.project_id)
258+
deployment_name: $(vars.deployment_name)
259+
region: $(vars.region)
260+
machine_type: a4-highgpu-8g
261+
gcs_bucket: $(vars.chs_output_bucket_name)
262+
gcs_pvc: $(vars.chs_pvc_claim_name)
263+
cronjob_schedule: $(vars.health_check_schedule)
234264
kueue:
235265
install: true
236266
config_path: $(vars.kueue_configuration_path)

0 commit comments

Comments
 (0)