-
-
Notifications
You must be signed in to change notification settings - Fork 4.3k
Expand file tree
/
Copy pathchanged-backends.js
More file actions
310 lines (284 loc) · 13.2 KB
/
Copy pathchanged-backends.js
File metadata and controls
310 lines (284 loc) · 13.2 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
import fs from "fs";
import * as yaml from "js-yaml";
import { Octokit } from "@octokit/core";
import {
getAllBackendPaths,
filterMatrix,
BACKEND_MATRIX_FILE,
} from "./lib/backend-filter.mjs";
// Matrix data lives in a small data-only YAML so both backend.yml (master push)
// and backend_pr.yml (pull_request) can use a dynamic `matrix: ${{ fromJson(...) }}`
// for the live job, while this script remains the single source of truth for
// "what backends does the project know about".
const matrixYml = yaml.load(fs.readFileSync(BACKEND_MATRIX_FILE, "utf8"));
const includes = matrixYml.include;
const includesDarwin = matrixYml.includeDarwin;
const eventPath = process.env.GITHUB_EVENT_PATH;
const event = JSON.parse(fs.readFileSync(eventPath, "utf8"));
const allBackendPaths = getAllBackendPaths(includes, includesDarwin);
const token = process.env.GITHUB_TOKEN;
const octokit = new Octokit({ auth: token });
// PR file list — paginated.
async function getChangedFilesForPR(event) {
const prNumber = event.pull_request.number;
const repo = event.repository.name;
const owner = event.repository.owner.login;
let files = [];
let page = 1;
while (true) {
const res = await octokit.request('GET /repos/{owner}/{repo}/pulls/{pull_number}/files', {
owner,
repo,
pull_number: prNumber,
per_page: 100,
page
});
files = files.concat(res.data.map(f => f.filename));
if (res.data.length < 100) break;
page++;
}
return files;
}
// Branch-push file list — uses the Compare API so it works in shallow clones.
// Returns null to signal "we cannot compute a reliable diff; run everything".
async function getChangedFilesForPush(event) {
const before = event.before;
const after = event.after;
// First push to a branch carries an all-zero `before` SHA and there's no
// base to diff against. Run everything in that case.
if (!before || !after || /^0+$/.test(before)) return null;
const owner = event.repository.owner.login;
const repo = event.repository.name;
let res;
try {
res = await octokit.request('GET /repos/{owner}/{repo}/compare/{basehead}', {
owner,
repo,
basehead: `${before}...${after}`,
});
} catch (err) {
console.log("compare API failed, falling back to run-all:", err.message);
return null;
}
if (!res.data || !Array.isArray(res.data.files)) return null;
// The compare endpoint caps the file list at 300. If we hit the cap we may
// be missing changes — be conservative and run everything.
if (res.data.files.length >= 300) {
console.log("compare API returned 300+ files (truncated), falling back to run-all");
return null;
}
return res.data.files.map(f => f.filename);
}
// The matrix file's contents at the base revision, so filterMatrix can rebuild
// only the entries whose fields actually changed instead of all 417 (or, as
// before, none of them). Returns null when the previous revision cannot be
// resolved, which filterMatrix treats as "rebuild everything".
//
// Only called when the changed-file list actually names the matrix file, so the
// common path costs no extra API request.
async function getPreviousMatrix(event) {
const ref = event.pull_request ? event.pull_request.base.sha : event.before;
if (!ref || /^0+$/.test(ref)) return null;
const owner = event.repository.owner.login;
const repo = event.repository.name;
try {
const res = await octokit.request('GET /repos/{owner}/{repo}/contents/{path}', {
owner,
repo,
path: BACKEND_MATRIX_FILE,
ref,
mediaType: { format: 'raw' },
});
// `format: raw` yields a string; fall back to the JSON representation in
// case a proxy or a future Octokit version ignores the media type.
const raw = typeof res.data === 'string'
? res.data
: Buffer.from(res.data.content, 'base64').toString('utf8');
const previous = yaml.load(raw);
return {
include: previous.include || [],
includeDarwin: previous.includeDarwin || [],
};
} catch (err) {
console.log(
`could not read ${BACKEND_MATRIX_FILE} at ${ref}, falling back to run-all:`,
err.message
);
return null;
}
}
// Group matrix entries by tag-suffix and emit a merge-matrix entry per group.
// Both multi-leg groups (per-arch fan-out) and singletons get one entry each:
// the build job pushes by digest only with no tags applied, so every backend
// needs a downstream merge step to apply its tags via `imagetools create`,
// regardless of how many per-arch legs feed it. Callers split entries by
// arch class first (see splitByArch) and call this once per class so the
// resulting matrices can be wired to merge jobs that `needs:` only their
// corresponding build matrix — preventing slow single-arch builds from
// gating multi-arch merges (the bug fixed in PR #9746).
function computeMergeMatrix(entries) {
const groups = new Map();
for (const item of entries) {
if (!item['tag-suffix']) continue;
const key = item['tag-suffix'];
if (!groups.has(key)) groups.set(key, []);
groups.get(key).push(item);
}
const include = [];
for (const [tagSuffix, group] of groups) {
// tag-latest must agree across legs — they're going to publish under
// the same final tag, so disagreeing on whether it's also the :latest
// tag is an authoring bug. Warn loudly so a Task 2.5 fan-out typo is
// visible in CI logs instead of silently shipping the leg-0 value.
const first = group[0]['tag-latest'] || '';
for (const m of group) {
if ((m['tag-latest'] || '') !== first) {
console.warn(`tag-latest mismatch in group ${tagSuffix}: legs disagree (using ${first})`);
break;
}
}
include.push({
'tag-suffix': tagSuffix,
'tag-latest': first,
});
}
return { include };
}
// Split a list of linux matrix entries into single-arch (no platform-tag) and
// multi-arch (platform-tag set, paired with a sibling entry sharing the same
// tag-suffix). The two are run as separate matrix jobs so backend-merge-jobs
// can `needs:` only the multi-arch one — slow single-arch builds (CUDA, ROCm,
// vLLM, etc.) don't block manifest assembly while their per-arch counterparts'
// untagged digests sit on quay long enough to be GC'd.
function splitByArch(entries) {
const multiarch = entries.filter(e => e['platform-tag']);
const singlearch = entries.filter(e => !e['platform-tag']);
return { multiarch, singlearch };
}
// GitHub Actions refuses to instantiate a matrix with more than 256 jobs. When
// it happens the job doesn't error visibly — it hangs forever at "Waiting for
// pending jobs" and the whole run is marked `failure` while every *other* job
// stays green (seen on the v4.6.1 tag build, run 28786533892: 268 single-arch
// entries, zero single-arch jobs ever created). The single-arch list is the
// one that grows unbounded as backends are added, so we shard it across a
// fixed number of matrix jobs instead of feeding one oversized matrix.
//
// SINGLEARCH_SHARDS MUST equal the number of backend-jobs-singlearch-<n>
// (and backend-merge-jobs-singlearch-<n>) blocks defined in backend.yml and
// backend_pr.yml. Bump all three together.
const SINGLEARCH_SHARDS = 4;
const GHA_MATRIX_LIMIT = 256;
// Split `arr` into exactly `shards` balanced, contiguous chunks. Earlier chunks
// absorb the remainder when the length doesn't divide evenly; trailing chunks
// may be empty when there are fewer entries than shards (those emit a
// has-backends-singlearch-<n>=false flag so their job is skipped).
function chunkEqually(arr, shards) {
const out = [];
const base = Math.floor(arr.length / shards);
const rem = arr.length % shards;
let idx = 0;
for (let i = 0; i < shards; i++) {
const size = base + (i < rem ? 1 : 0);
out.push(arr.slice(idx, idx + size));
idx += size;
}
return out;
}
// Emit the sharded single-arch build + merge matrices and their has-* gates.
// Called with the full or filtered single-arch entry list.
function emitSinglearchShards(singlearch) {
const shards = chunkEqually(singlearch, SINGLEARCH_SHARDS);
for (let i = 0; i < SINGLEARCH_SHARDS; i++) {
const shard = shards[i];
// Fail loudly rather than let GitHub silently drop the overflow: a shard at
// or above the limit means SINGLEARCH_SHARDS (and the matching job blocks in
// both workflows) need to grow.
if (shard.length >= GHA_MATRIX_LIMIT) {
throw new Error(
`single-arch shard ${i + 1} has ${shard.length} entries (>= ${GHA_MATRIX_LIMIT}, ` +
`GitHub's per-matrix job limit). Increase SINGLEARCH_SHARDS in ` +
`scripts/changed-backends.js and add matching backend-jobs-singlearch-<n> / ` +
`backend-merge-jobs-singlearch-<n> blocks to backend.yml and backend_pr.yml.`
);
}
const merge = computeMergeMatrix(shard);
const n = i + 1;
fs.appendFileSync(process.env.GITHUB_OUTPUT, `has-backends-singlearch-${n}=${shard.length > 0 ? 'true' : 'false'}\n`);
fs.appendFileSync(process.env.GITHUB_OUTPUT, `has-merges-singlearch-${n}=${merge.include.length > 0 ? 'true' : 'false'}\n`);
fs.appendFileSync(process.env.GITHUB_OUTPUT, `matrix-singlearch-${n}=${JSON.stringify({ include: shard })}\n`);
fs.appendFileSync(process.env.GITHUB_OUTPUT, `merge-matrix-singlearch-${n}=${JSON.stringify(merge)}\n`);
}
}
function emitFullMatrix() {
const { multiarch, singlearch } = splitByArch(includes);
const mergeMatrixMultiarch = computeMergeMatrix(multiarch);
const hasMergesMultiarch = mergeMatrixMultiarch.include.length > 0 ? 'true' : 'false';
fs.appendFileSync(process.env.GITHUB_OUTPUT, `run-all=true\n`);
fs.appendFileSync(process.env.GITHUB_OUTPUT, `has-backends-multiarch=${multiarch.length > 0 ? 'true' : 'false'}\n`);
fs.appendFileSync(process.env.GITHUB_OUTPUT, `has-backends-darwin=true\n`);
fs.appendFileSync(process.env.GITHUB_OUTPUT, `has-merges-multiarch=${hasMergesMultiarch}\n`);
fs.appendFileSync(process.env.GITHUB_OUTPUT, `matrix-multiarch=${JSON.stringify({ include: multiarch })}\n`);
fs.appendFileSync(process.env.GITHUB_OUTPUT, `matrix-darwin=${JSON.stringify({ include: includesDarwin })}\n`);
fs.appendFileSync(process.env.GITHUB_OUTPUT, `merge-matrix-multiarch=${JSON.stringify(mergeMatrixMultiarch)}\n`);
emitSinglearchShards(singlearch);
for (const backend of allBackendPaths.keys()) {
fs.appendFileSync(process.env.GITHUB_OUTPUT, `${backend}=true\n`);
}
}
function emitFilteredMatrix(changedFiles, previousMatrix) {
console.log("Changed files:", changedFiles);
const { filtered, filteredDarwin, changedBackends } = filterMatrix({
includes,
includesDarwin,
changedFiles,
previousMatrix,
});
console.log("Filtered files:", filtered);
console.log("Filtered files Darwin:", filteredDarwin);
const { multiarch, singlearch } = splitByArch(filtered);
const hasBackendsMultiarch = multiarch.length > 0 ? 'true' : 'false';
const hasBackendsDarwin = filteredDarwin.length > 0 ? 'true' : 'false';
console.log("Has single-arch backends?:", singlearch.length > 0 ? 'true' : 'false');
console.log("Has multi-arch backends?:", hasBackendsMultiarch);
console.log("Has Darwin backends?:", hasBackendsDarwin);
const mergeMatrixMultiarch = computeMergeMatrix(multiarch);
const hasMergesMultiarch = mergeMatrixMultiarch.include.length > 0 ? 'true' : 'false';
fs.appendFileSync(process.env.GITHUB_OUTPUT, `run-all=false\n`);
fs.appendFileSync(process.env.GITHUB_OUTPUT, `has-backends-multiarch=${hasBackendsMultiarch}\n`);
fs.appendFileSync(process.env.GITHUB_OUTPUT, `has-backends-darwin=${hasBackendsDarwin}\n`);
fs.appendFileSync(process.env.GITHUB_OUTPUT, `has-merges-multiarch=${hasMergesMultiarch}\n`);
fs.appendFileSync(process.env.GITHUB_OUTPUT, `matrix-multiarch=${JSON.stringify({ include: multiarch })}\n`);
fs.appendFileSync(process.env.GITHUB_OUTPUT, `matrix-darwin=${JSON.stringify({ include: filteredDarwin })}\n`);
fs.appendFileSync(process.env.GITHUB_OUTPUT, `merge-matrix-multiarch=${JSON.stringify(mergeMatrixMultiarch)}\n`);
emitSinglearchShards(singlearch);
// Per-backend boolean outputs
for (const backend of allBackendPaths.keys()) {
const changed = changedBackends.has(backend);
fs.appendFileSync(process.env.GITHUB_OUTPUT, `${backend}=${changed ? 'true' : 'false'}\n`);
}
}
(async () => {
// Tag pushes and an explicit FORCE_ALL escape hatch always rebuild everything.
// FORCE_ALL is set from backend.yml whenever github.ref starts with refs/tags/.
const forceAll = process.env.FORCE_ALL === 'true';
const isTagPush = typeof event.ref === 'string' && event.ref.startsWith('refs/tags/');
const isBranchPush = !!event.ref && !event.pull_request && !isTagPush;
let changedFiles = null;
if (event.pull_request) {
changedFiles = await getChangedFilesForPR(event);
} else if (isBranchPush && !forceAll) {
changedFiles = await getChangedFilesForPush(event);
// null -> fall through to the full matrix (e.g. first push, API truncated,
// network failure).
}
// All other event types (workflow_dispatch, schedule, tag pushes, FORCE_ALL)
// leave changedFiles === null and run everything.
if (changedFiles === null) {
emitFullMatrix();
return;
}
const previousMatrix = changedFiles.includes(BACKEND_MATRIX_FILE)
? await getPreviousMatrix(event)
: null;
emitFilteredMatrix(changedFiles, previousMatrix);
})();