-
Notifications
You must be signed in to change notification settings - Fork 3
Expand file tree
/
Copy pathfleet-health-status.yml
More file actions
343 lines (320 loc) · 16 KB
/
Copy pathfleet-health-status.yml
File metadata and controls
343 lines (320 loc) · 16 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
# AZLOCAL-PIPELINE-ID: fleet-health-status
# Fleet Health Status Monitoring
# This workflow surfaces 24-hour system health-check failures across every Azure Local
# cluster the service principal can read - including clusters that are already
# "up to date" with no available updates. The 24-hour health checks continue to run
# on the cluster even when no update is in flight, so this pipeline is the dedicated
# place to triage fleet-wide health issues that exist OUTSIDE the update workflow.
#
# USE CASES:
# - Daily / weekly fleet-wide critical & warning health audit
# - Executive dashboard: "which failure reason hits the most clusters?"
# - Compliance tracking for cluster health, independent of update activity
# - Triage hand-off into ITSM: the JUnit XML and CSV exports are ready-made
# incident payloads
#
# REPORTS GENERATED:
# - Markdown step summary: pivoted by FailureReason (top failures by cluster impact),
# matching the "Health Check Failures By Reason" view administrators are used to,
# and including a per-cluster "Detailed Results" table further down.
# - JUnit XML (diagnostic mirror): one <testcase time="0"> per (cluster, failing health check).
# dorny test-reporter renders failed Critical checks as <failure> and Warning
# checks as <failure type="Warning"> so reviewers can drill from suite ->
# cluster -> failure.
# - CSV (detail): fleet-health-detail.csv - one row per (cluster, failing check)
# - CSV (summary): fleet-health-summary.csv - one row per (FailureReason, Severity)
# - JSON: same shape, machine-readable for downstream automation
#
# AUTHENTICATION:
# Uses OpenID Connect (OIDC) - recommended for secretless authentication
# See: https://learn.microsoft.com/en-us/azure/developer/github/connect-from-azure-openid-connect
# Workflow name carries the same Step.N - prefix as the filename so the GitHub
# Actions sidebar (which sorts workflows alphabetically by this `name:` field)
# lists the eight pipelines in execution order.
name: Step.10 - Fleet Health Status
on:
# BEGIN-AZLOCAL-CUSTOMIZE:schedule-triggers
# Edits inside this block are preserved across Update-AzLocalPipelineExample
# module upgrades. Add or modify `schedule:` cron blocks below.
# Run on schedule (e.g., daily at 7 AM UTC, offset from Fleet Update Status so the two
# do not start their CLI shell-outs in the same minute).
schedule:
- cron: '0 7 * * *' # Daily at 7:00 AM UTC
# END-AZLOCAL-CUSTOMIZE:schedule-triggers
# Manual trigger with options
workflow_dispatch:
inputs:
scope:
description: 'Scope of clusters to check'
required: true
default: 'all'
type: choice
options:
- 'all' # All clusters across all subscriptions
- 'by-update-ring' # Filter by UpdateRing tag
update_ring:
# accepts a single ring (Wave1), a semicolon-delimited list (Prod;Ring2),
# or '***' (three stars - deliberate, not a typo) to match every cluster that HAS the UpdateRing tag set.
description: "UpdateRing tag value (only used when scope=by-update-ring). Single ring, 'Prod;Ring2', or '***'."
required: false
default: ''
severity:
description: 'Severity filter applied at Resource Graph (default All = Critical + Warning; Informational is always excluded).'
required: false
default: 'All'
type: choice
options:
- 'All'
- 'Critical'
- 'Warning'
module_version:
description: 'Pin AzLocal.UpdateManagement version (empty = latest from PSGallery). See Automation-Pipeline-Examples/README.md section 5 "Optional configuration".'
required: false
default: ''
# --- ITSM Connector (ServiceNow auto-raise on fleet-health failures) ---
# Set raise_itsm_ticket=true to open ServiceNow incidents from each Critical /
# Warning health-check failure published by this pipeline. Default is false
# so existing schedules stay byte-identical until you opt in. The connector reads
# the JUnit file this pipeline already produces (./reports/fleet-health-status.xml) and
# the trigger matrix in ./.itsm/azurelocal-itsm.yml. Dedupe granularity is one
# ticket per (cluster, failing health check) pair - FailureReason is fed into
# the UpdateName slot of the SHA256 dedupe key.
raise_itsm_ticket:
description: 'Open ITSM tickets (ServiceNow) for fleet-health failures'
required: false
default: 'false'
type: choice
options:
- 'false'
- 'true'
itsm_config_path:
description: 'Path to ITSM matrix config (YAML or JSON)'
required: false
default: './.itsm/azurelocal-itsm.yml'
itsm_dry_run:
description: 'ITSM: build payloads + run read-only dedupe but do NOT create tickets'
required: false
default: 'false'
type: choice
options:
- 'false'
- 'true'
itsm_force_create:
description: 'ITSM: bypass dedupe and always create new tickets (use with caution)'
required: false
default: 'false'
type: choice
options:
- 'false'
- 'true'
env:
# Module version this workflow YAML was generated against. The install step compares
# this to the version actually installed and to the latest on PSGallery, and emits a
# ::notice annotation if the YAML appears stale - prompting you to refresh via
# Copy-AzLocalPipelineExample -Update. See Automation-Pipeline-Examples/README.md section 5.
GENERATED_AGAINST_MODULE_VERSION: '0.8.77'
# Resolution order for the module version pin (leave all unset to install the latest,
# which is the default "fix-forward" behaviour): manual workflow_dispatch input >
# repository variable 'REQUIRED_MODULE_VERSION' > empty (latest).
REQUIRED_MODULE_VERSION: ${{ github.event.inputs.module_version || vars.REQUIRED_MODULE_VERSION || '' }}
# v0.8.4 - opt this workflow into Node.js 24 for all JavaScript actions
# (actions/checkout, actions/download-artifact, actions/upload-artifact,
# azure/login, dorny/test-reporter, etc). Per GitHub's 2025-09-19 deprecation
# notice Node 20 is forced off by default on 2026-06-16 and removed from the
# runner on 2026-09-16. Setting this env var silences the deprecation warnings
# and exercises Node 24 ahead of the cut-over. To temporarily opt back out,
# set ACTIONS_ALLOW_USE_UNSECURE_NODE_VERSION=true.
# https://github.blog/changelog/2025-09-19-deprecation-of-node-20-on-github-actions-runners/
FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: true
jobs:
fleet-health:
name: Collect Fleet Health Status
runs-on: windows-latest
permissions:
id-token: write
contents: read
checks: write # Required for test reporter
steps:
- name: Checkout repository
uses: actions/checkout@v5
# OIDC Authentication - recommended
- name: Azure CLI Login (OIDC)
uses: azure/login@v3
with:
client-id: ${{ secrets.AZURE_CLIENT_ID }}
tenant-id: ${{ vars.AZURE_TENANT_ID }}
# AZURE_SUBSCRIPTION_ID is a repository *Variable* (vars.*), not a Secret. It
# is consumed ONLY here: azure/login@v3 runs `az account set --subscription
# <id>` after the OIDC token exchange so the runner has a default
# `az account` context. It is NOT used to scope Azure Resource Graph queries
# (those run fleet-wide across every subscription the federated identity can
# read) and is NOT interpolated into Azure portal deep-link URLs (those use
# the per-row `subscriptionId` returned by ARG).
# Set it via: gh variable set AZURE_SUBSCRIPTION_ID --body <subId>
subscription-id: ${{ vars.AZURE_SUBSCRIPTION_ID }}
- name: Install Azure CLI Resource Graph Extension
shell: pwsh
run: |
az extension add --name resource-graph --yes
- name: Install AzLocal.UpdateManagement from PSGallery
# v0.8.5 thin-YAML: drift detection + banner + step outputs are all
# produced by Add-AzLocalPipelineVersionBanner (Public cmdlet).
shell: pwsh
id: module-version
run: |
$ErrorActionPreference = 'Stop'
$installArgs = @{ Name = 'AzLocal.UpdateManagement'; Scope = 'CurrentUser'; Force = $true; AllowClobber = $true }
if ($env:REQUIRED_MODULE_VERSION) {
$installArgs.RequiredVersion = $env:REQUIRED_MODULE_VERSION
Write-Host "REQUIRED_MODULE_VERSION is set - pinning install to v$($env:REQUIRED_MODULE_VERSION)."
} else {
Write-Host "REQUIRED_MODULE_VERSION is empty - installing the latest version from PSGallery (default fix-forward behaviour)."
}
Install-Module @installArgs
Import-Module AzLocal.UpdateManagement -Force
Add-AzLocalPipelineVersionBanner `
-GeneratedAgainstVersion $env:GENERATED_AGAINST_MODULE_VERSION `
-PinnedVersion $env:REQUIRED_MODULE_VERSION
- name: Collect Fleet Health Status
# v0.8.5 thin-YAML: the inline run block (Get-AzLocalFleetHealthFailures
# Detail view + in-process Group-Object summary roll-up +
# Get-AzLocalFleetHealthOverview + 2-suite JUnit XML emission +
# 4-section markdown step summary written to GITHUB_STEP_SUMMARY +
# 8 step outputs) has been condensed into the Public cmdlet
# Export-AzLocalFleetHealthStatusReport. The cmdlet writes
# ./reports/*.{json,csv,xml} for the upload-artifact step and sets the
# 8 step outputs consumed by downstream jobs (e.g. ITSM connector).
id: fleet-health
shell: pwsh
env:
INPUT_SCOPE: ${{ github.event.inputs.scope || 'all' }}
INPUT_UPDATE_RING: ${{ github.event.inputs.update_ring }}
INPUT_SEVERITY: ${{ github.event.inputs.severity || 'All' }}
INSTALLED_MODULE_VERSION: ${{ steps.module-version.outputs.installed_module_version }}
run: |
$ErrorActionPreference = 'Stop'
Import-Module AzLocal.UpdateManagement -Force
$params = @{
Scope = if ($env:INPUT_SCOPE) { $env:INPUT_SCOPE } else { 'all' }
Severity = if ($env:INPUT_SEVERITY) { $env:INPUT_SEVERITY } else { 'All' }
InstalledModuleVersion = $env:INSTALLED_MODULE_VERSION
}
if ($env:INPUT_UPDATE_RING) { $params['UpdateRing'] = $env:INPUT_UPDATE_RING }
Export-AzLocalFleetHealthStatusReport @params
- name: Compute Artifact Timestamp
if: always()
id: artifact-stamp
shell: pwsh
# every downloadable artifact gets a UTC timestamp suffix so multiple runs on
# the same day produce distinct zip names.
run: |
$stamp = (Get-Date).ToUniversalTime().ToString('yyyyMMdd_HHmmss')
"timestamp=$stamp" | Out-File -FilePath $env:GITHUB_OUTPUT -Encoding utf8 -Append
Write-Host "Artifact timestamp: $stamp"
- name: Upload Fleet Health Reports
uses: actions/upload-artifact@v6
with:
name: azlocal-step.10-fleet-health-status-report_${{ steps.artifact-stamp.outputs.timestamp }}
path: ./reports/
retention-days: 90
# Create the markdown summary FIRST so it appears at the top of the GitHub
# Actions run summary view. Publish Test Results (dorny) runs AFTER and is
# configured with list-suites/list-tests=failed so the testsuite expansion
# stays compact for large fleets (only failing rows are rendered).
- name: Publish JUnit Diagnostic Results
uses: dorny/test-reporter@v3
if: always()
with:
name: Fleet Health Status JUnit Debug
path: ./reports/fleet-health-status.xml
reporter: java-junit
fail-on-error: false # Don't fail the action on health failures
list-suites: failed # collapse passing suites (compact view for large fleets)
list-tests: failed # within failed suites, only render failed tests
# ----------------------------------------------------------------------
# ITSM Connector (ServiceNow auto-raise on fleet-health failures)
#
# Fully opt-in (gated on inputs.raise_itsm_ticket == 'true'). Runs AFTER
# the JUnit file is uploaded + test results published, so a) failing
# checks are already surfaced via dorny/test-reporter and the run
# summary, and b) the ITSM action is strictly additive - failures here
# never affect the fleet-health-status exit code.
#
# Reads the same fleet-health-status.xml that powers the run summary.
# The JUnit emitter writes per-testcase <properties> (ClusterName /
# ClusterResourceId / UpdateName=FailureReason / Status=Severity /
# FailureReason / Severity / ClusterPortalUrl / TargetResourceName /
# TargetResourceType) so New-AzLocalIncident can compute the SHA256
# dedupe key (one ticket per cluster + failing check) and the Mustache
# body template can deep-link straight into the cluster blade.
# ----------------------------------------------------------------------
- name: Install powershell-yaml (ITSM config parser)
if: ${{ github.event.inputs.raise_itsm_ticket == 'true' }}
shell: pwsh
run: |
if (-not (Get-Module -ListAvailable -Name powershell-yaml)) {
Install-Module powershell-yaml -Scope CurrentUser -Force -AllowClobber
}
- name: Raise ITSM tickets
if: ${{ github.event.inputs.raise_itsm_ticket == 'true' }}
shell: pwsh
id: itsm
env:
# BEGIN-AZLOCAL-CUSTOMIZE:itsm-secrets
# Bind your ITSM connector secrets here. Defaults match the
# ServiceNow OAuth client_credentials naming used by azurelocal-itsm.yml.
# Preserved by Update-AzLocalPipelineExample across module upgrades.
ITSM_SN_INSTANCE_URL: ${{ secrets.ITSM_SN_INSTANCE_URL }}
ITSM_SN_CLIENT_ID: ${{ secrets.ITSM_SN_CLIENT_ID }}
ITSM_SN_CLIENT_SECRET: ${{ secrets.ITSM_SN_CLIENT_SECRET }}
# END-AZLOCAL-CUSTOMIZE:itsm-secrets
INPUT_ITSM_CONFIG_PATH: ${{ github.event.inputs.itsm_config_path }}
INPUT_ITSM_DRY_RUN: ${{ github.event.inputs.itsm_dry_run }}
INPUT_ITSM_FORCE_CREATE: ${{ github.event.inputs.itsm_force_create }}
run: |
Import-Module AzLocal.UpdateManagement -Force
$configPath = $env:INPUT_ITSM_CONFIG_PATH
$dryRun = $env:INPUT_ITSM_DRY_RUN -eq 'true'
$force = $env:INPUT_ITSM_FORCE_CREATE -eq 'true'
if (-not (Test-Path $configPath)) {
Write-Host "::warning::ITSM config not found at '$configPath' - skipping ticket creation."
exit 0
}
$cfg = Get-AzLocalItsmConfig -Path $configPath
$junitInput = './reports/fleet-health-status.xml'
if (-not (Test-Path $junitInput)) {
Write-Host "::warning::No fleet-health-status.xml found at '$junitInput' - skipping ticket creation."
exit 0
}
$params = @{
InputArtifactPath = $junitInput
Config = $cfg
RunMetadata = @{
Platform = 'github'
RunId = $env:GITHUB_RUN_ID
RunUrl = "$env:GITHUB_SERVER_URL/$env:GITHUB_REPOSITORY/actions/runs/$env:GITHUB_RUN_ID"
Branch = $env:GITHUB_REF
}
DryRun = $dryRun
ForceCreate = $force
ExportPath = './reports/itsm-results.csv'
ExportJUnitPath = './reports/itsm-results.xml'
}
$results = New-AzLocalIncident @params
$results | Format-Table ClusterName, Action, TicketId, Severity -AutoSize
- name: Upload ITSM Artefacts
if: ${{ github.event.inputs.raise_itsm_ticket == 'true' }}
uses: actions/upload-artifact@v6
with:
name: azlocal-step.10-fleet-health-status-itsm-results_${{ steps.artifact-stamp.outputs.timestamp }}
path: ./reports/itsm-*.*
retention-days: 30
continue-on-error: true
- name: Publish ITSM Test Results
if: ${{ github.event.inputs.raise_itsm_ticket == 'true' }}
uses: dorny/test-reporter@v3
with:
name: ITSM Tickets (Step.10)
path: ./reports/itsm-results.xml
reporter: java-junit
continue-on-error: true