mirror of
https://github.com/mudler/LocalAI.git
synced 2026-07-30 18:09:05 -04:00
backend/backend.proto is consumed by every language, so its SHARED_BUILD_INPUTS rule could only ever be always/always: 417 Linux plus 56 Darwin builds. It fires on ~1.3% of commits (10 of 767 over six months), which made it the single largest CI cost driver in the repo. On 2026-07-29 the queue reached 2178 jobs against 8 concurrent runners. Four runs totalling 935 of those jobs were triggered by nothing but a proto edit. The largest, 378 jobs on master, came from PR #11158, whose entire proto diff was six lines adding `bool cache_prompt = 8;` to one message. No backend that does not read that field behaves any differently for it. Make the rule content-aware. changed-backends.js resolves backend.proto at the base revision (the contents-API pattern already used for backend-matrix.yml) and hands both texts to protoChangeIsAdditive(), which compares them structurally so a comment reflow, reindent or field reorder does not read as a change. An additive-only edit (new field with an unused number, new message, new enum value, new RPC) suppresses the rule and rebuilds nothing; a removed, renumbered, retyped or renamed field, a dropped RPC or a changed option still rebuilds everything, as does an unresolvable base revision. Every other matched rule is untouched, so a PR that edits the proto and scripts/build/ is still a full rebuild, and the weekly full-matrix cron remains the backstop for stale wheels. Verified against all ten proto commits of the preceding six months: the nine with a resolvable parent all classify as additive, and controls covering a retyped-and-renumbered field, a deleted RPC, identical revisions and a reindent-plus-comment-reflow all classify correctly. Assisted-by: Claude:opus-5 [claude-code] Signed-off-by: Ettore Di Giacinto <mudler@localai.io> Co-authored-by: Ettore Di Giacinto <mudler@localai.io>
353 lines
15 KiB
JavaScript
353 lines
15 KiB
JavaScript
import fs from "fs";
|
|
import * as yaml from "js-yaml";
|
|
import { Octokit } from "@octokit/core";
|
|
|
|
import {
|
|
getAllBackendPaths,
|
|
filterMatrix,
|
|
BACKEND_MATRIX_FILE,
|
|
BACKEND_PROTO_FILE,
|
|
} from "./lib/backend-filter.mjs";
|
|
|
|
// Matrix data lives in a small data-only YAML so both backend.yml (master push)
|
|
// and backend_pr.yml (pull_request) can use a dynamic `matrix: ${{ fromJson(...) }}`
|
|
// for the live job, while this script remains the single source of truth for
|
|
// "what backends does the project know about".
|
|
const matrixYml = yaml.load(fs.readFileSync(BACKEND_MATRIX_FILE, "utf8"));
|
|
const includes = matrixYml.include;
|
|
const includesDarwin = matrixYml.includeDarwin;
|
|
|
|
const eventPath = process.env.GITHUB_EVENT_PATH;
|
|
const event = JSON.parse(fs.readFileSync(eventPath, "utf8"));
|
|
|
|
const allBackendPaths = getAllBackendPaths(includes, includesDarwin);
|
|
|
|
const token = process.env.GITHUB_TOKEN;
|
|
const octokit = new Octokit({ auth: token });
|
|
|
|
// PR file list — paginated.
|
|
async function getChangedFilesForPR(event) {
|
|
const prNumber = event.pull_request.number;
|
|
const repo = event.repository.name;
|
|
const owner = event.repository.owner.login;
|
|
let files = [];
|
|
let page = 1;
|
|
while (true) {
|
|
const res = await octokit.request('GET /repos/{owner}/{repo}/pulls/{pull_number}/files', {
|
|
owner,
|
|
repo,
|
|
pull_number: prNumber,
|
|
per_page: 100,
|
|
page
|
|
});
|
|
files = files.concat(res.data.map(f => f.filename));
|
|
if (res.data.length < 100) break;
|
|
page++;
|
|
}
|
|
return files;
|
|
}
|
|
|
|
// Branch-push file list — uses the Compare API so it works in shallow clones.
|
|
// Returns null to signal "we cannot compute a reliable diff; run everything".
|
|
async function getChangedFilesForPush(event) {
|
|
const before = event.before;
|
|
const after = event.after;
|
|
// First push to a branch carries an all-zero `before` SHA and there's no
|
|
// base to diff against. Run everything in that case.
|
|
if (!before || !after || /^0+$/.test(before)) return null;
|
|
const owner = event.repository.owner.login;
|
|
const repo = event.repository.name;
|
|
let res;
|
|
try {
|
|
res = await octokit.request('GET /repos/{owner}/{repo}/compare/{basehead}', {
|
|
owner,
|
|
repo,
|
|
basehead: `${before}...${after}`,
|
|
});
|
|
} catch (err) {
|
|
console.log("compare API failed, falling back to run-all:", err.message);
|
|
return null;
|
|
}
|
|
if (!res.data || !Array.isArray(res.data.files)) return null;
|
|
// The compare endpoint caps the file list at 300. If we hit the cap we may
|
|
// be missing changes — be conservative and run everything.
|
|
if (res.data.files.length >= 300) {
|
|
console.log("compare API returned 300+ files (truncated), falling back to run-all");
|
|
return null;
|
|
}
|
|
return res.data.files.map(f => f.filename);
|
|
}
|
|
|
|
// The matrix file's contents at the base revision, so filterMatrix can rebuild
|
|
// only the entries whose fields actually changed instead of all 417 (or, as
|
|
// before, none of them). Returns null when the previous revision cannot be
|
|
// resolved, which filterMatrix treats as "rebuild everything".
|
|
//
|
|
// Only called when the changed-file list actually names the matrix file, so the
|
|
// common path costs no extra API request.
|
|
async function getPreviousMatrix(event) {
|
|
const ref = event.pull_request ? event.pull_request.base.sha : event.before;
|
|
if (!ref || /^0+$/.test(ref)) return null;
|
|
const owner = event.repository.owner.login;
|
|
const repo = event.repository.name;
|
|
try {
|
|
const res = await octokit.request('GET /repos/{owner}/{repo}/contents/{path}', {
|
|
owner,
|
|
repo,
|
|
path: BACKEND_MATRIX_FILE,
|
|
ref,
|
|
mediaType: { format: 'raw' },
|
|
});
|
|
// `format: raw` yields a string; fall back to the JSON representation in
|
|
// case a proxy or a future Octokit version ignores the media type.
|
|
const raw = typeof res.data === 'string'
|
|
? res.data
|
|
: Buffer.from(res.data.content, 'base64').toString('utf8');
|
|
const previous = yaml.load(raw);
|
|
return {
|
|
include: previous.include || [],
|
|
includeDarwin: previous.includeDarwin || [],
|
|
};
|
|
} catch (err) {
|
|
console.log(
|
|
`could not read ${BACKEND_MATRIX_FILE} at ${ref}, falling back to run-all:`,
|
|
err.message
|
|
);
|
|
return null;
|
|
}
|
|
}
|
|
|
|
// backend.proto at the base revision plus the checked-out copy, so filterMatrix
|
|
// can tell an additive edit (a new field, message or RPC, which invalidates no
|
|
// existing image) from a breaking one. Returning null means "rebuild
|
|
// everything", the same posture as an unresolvable matrix diff.
|
|
//
|
|
// Only called when the changed-file list names the proto, so the common path
|
|
// costs no extra API request.
|
|
async function getProtoRevisions(event) {
|
|
const ref = event.pull_request ? event.pull_request.base.sha : event.before;
|
|
if (!ref || /^0+$/.test(ref)) return null;
|
|
const owner = event.repository.owner.login;
|
|
const repo = event.repository.name;
|
|
try {
|
|
const res = await octokit.request('GET /repos/{owner}/{repo}/contents/{path}', {
|
|
owner,
|
|
repo,
|
|
path: BACKEND_PROTO_FILE,
|
|
ref,
|
|
mediaType: { format: 'raw' },
|
|
});
|
|
const previous = typeof res.data === 'string'
|
|
? res.data
|
|
: Buffer.from(res.data.content, 'base64').toString('utf8');
|
|
return {
|
|
previous,
|
|
current: fs.readFileSync(BACKEND_PROTO_FILE, "utf8"),
|
|
};
|
|
} catch (err) {
|
|
console.log(
|
|
`could not read ${BACKEND_PROTO_FILE} at ${ref}, falling back to run-all:`,
|
|
err.message
|
|
);
|
|
return null;
|
|
}
|
|
}
|
|
|
|
// Group matrix entries by tag-suffix and emit a merge-matrix entry per group.
|
|
// Both multi-leg groups (per-arch fan-out) and singletons get one entry each:
|
|
// the build job pushes by digest only with no tags applied, so every backend
|
|
// needs a downstream merge step to apply its tags via `imagetools create`,
|
|
// regardless of how many per-arch legs feed it. Callers split entries by
|
|
// arch class first (see splitByArch) and call this once per class so the
|
|
// resulting matrices can be wired to merge jobs that `needs:` only their
|
|
// corresponding build matrix — preventing slow single-arch builds from
|
|
// gating multi-arch merges (the bug fixed in PR #9746).
|
|
function computeMergeMatrix(entries) {
|
|
const groups = new Map();
|
|
for (const item of entries) {
|
|
if (!item['tag-suffix']) continue;
|
|
const key = item['tag-suffix'];
|
|
if (!groups.has(key)) groups.set(key, []);
|
|
groups.get(key).push(item);
|
|
}
|
|
const include = [];
|
|
for (const [tagSuffix, group] of groups) {
|
|
// tag-latest must agree across legs — they're going to publish under
|
|
// the same final tag, so disagreeing on whether it's also the :latest
|
|
// tag is an authoring bug. Warn loudly so a Task 2.5 fan-out typo is
|
|
// visible in CI logs instead of silently shipping the leg-0 value.
|
|
const first = group[0]['tag-latest'] || '';
|
|
for (const m of group) {
|
|
if ((m['tag-latest'] || '') !== first) {
|
|
console.warn(`tag-latest mismatch in group ${tagSuffix}: legs disagree (using ${first})`);
|
|
break;
|
|
}
|
|
}
|
|
include.push({
|
|
'tag-suffix': tagSuffix,
|
|
'tag-latest': first,
|
|
});
|
|
}
|
|
return { include };
|
|
}
|
|
|
|
// Split a list of linux matrix entries into single-arch (no platform-tag) and
|
|
// multi-arch (platform-tag set, paired with a sibling entry sharing the same
|
|
// tag-suffix). The two are run as separate matrix jobs so backend-merge-jobs
|
|
// can `needs:` only the multi-arch one — slow single-arch builds (CUDA, ROCm,
|
|
// vLLM, etc.) don't block manifest assembly while their per-arch counterparts'
|
|
// untagged digests sit on quay long enough to be GC'd.
|
|
function splitByArch(entries) {
|
|
const multiarch = entries.filter(e => e['platform-tag']);
|
|
const singlearch = entries.filter(e => !e['platform-tag']);
|
|
return { multiarch, singlearch };
|
|
}
|
|
|
|
// GitHub Actions refuses to instantiate a matrix with more than 256 jobs. When
|
|
// it happens the job doesn't error visibly — it hangs forever at "Waiting for
|
|
// pending jobs" and the whole run is marked `failure` while every *other* job
|
|
// stays green (seen on the v4.6.1 tag build, run 28786533892: 268 single-arch
|
|
// entries, zero single-arch jobs ever created). The single-arch list is the
|
|
// one that grows unbounded as backends are added, so we shard it across a
|
|
// fixed number of matrix jobs instead of feeding one oversized matrix.
|
|
//
|
|
// SINGLEARCH_SHARDS MUST equal the number of backend-jobs-singlearch-<n>
|
|
// (and backend-merge-jobs-singlearch-<n>) blocks defined in backend.yml and
|
|
// backend_pr.yml. Bump all three together.
|
|
const SINGLEARCH_SHARDS = 4;
|
|
const GHA_MATRIX_LIMIT = 256;
|
|
|
|
// Split `arr` into exactly `shards` balanced, contiguous chunks. Earlier chunks
|
|
// absorb the remainder when the length doesn't divide evenly; trailing chunks
|
|
// may be empty when there are fewer entries than shards (those emit a
|
|
// has-backends-singlearch-<n>=false flag so their job is skipped).
|
|
function chunkEqually(arr, shards) {
|
|
const out = [];
|
|
const base = Math.floor(arr.length / shards);
|
|
const rem = arr.length % shards;
|
|
let idx = 0;
|
|
for (let i = 0; i < shards; i++) {
|
|
const size = base + (i < rem ? 1 : 0);
|
|
out.push(arr.slice(idx, idx + size));
|
|
idx += size;
|
|
}
|
|
return out;
|
|
}
|
|
|
|
// Emit the sharded single-arch build + merge matrices and their has-* gates.
|
|
// Called with the full or filtered single-arch entry list.
|
|
function emitSinglearchShards(singlearch) {
|
|
const shards = chunkEqually(singlearch, SINGLEARCH_SHARDS);
|
|
for (let i = 0; i < SINGLEARCH_SHARDS; i++) {
|
|
const shard = shards[i];
|
|
// Fail loudly rather than let GitHub silently drop the overflow: a shard at
|
|
// or above the limit means SINGLEARCH_SHARDS (and the matching job blocks in
|
|
// both workflows) need to grow.
|
|
if (shard.length >= GHA_MATRIX_LIMIT) {
|
|
throw new Error(
|
|
`single-arch shard ${i + 1} has ${shard.length} entries (>= ${GHA_MATRIX_LIMIT}, ` +
|
|
`GitHub's per-matrix job limit). Increase SINGLEARCH_SHARDS in ` +
|
|
`scripts/changed-backends.js and add matching backend-jobs-singlearch-<n> / ` +
|
|
`backend-merge-jobs-singlearch-<n> blocks to backend.yml and backend_pr.yml.`
|
|
);
|
|
}
|
|
const merge = computeMergeMatrix(shard);
|
|
const n = i + 1;
|
|
fs.appendFileSync(process.env.GITHUB_OUTPUT, `has-backends-singlearch-${n}=${shard.length > 0 ? 'true' : 'false'}\n`);
|
|
fs.appendFileSync(process.env.GITHUB_OUTPUT, `has-merges-singlearch-${n}=${merge.include.length > 0 ? 'true' : 'false'}\n`);
|
|
fs.appendFileSync(process.env.GITHUB_OUTPUT, `matrix-singlearch-${n}=${JSON.stringify({ include: shard })}\n`);
|
|
fs.appendFileSync(process.env.GITHUB_OUTPUT, `merge-matrix-singlearch-${n}=${JSON.stringify(merge)}\n`);
|
|
}
|
|
}
|
|
|
|
function emitFullMatrix() {
|
|
const { multiarch, singlearch } = splitByArch(includes);
|
|
const mergeMatrixMultiarch = computeMergeMatrix(multiarch);
|
|
const hasMergesMultiarch = mergeMatrixMultiarch.include.length > 0 ? 'true' : 'false';
|
|
fs.appendFileSync(process.env.GITHUB_OUTPUT, `run-all=true\n`);
|
|
fs.appendFileSync(process.env.GITHUB_OUTPUT, `has-backends-multiarch=${multiarch.length > 0 ? 'true' : 'false'}\n`);
|
|
fs.appendFileSync(process.env.GITHUB_OUTPUT, `has-backends-darwin=true\n`);
|
|
fs.appendFileSync(process.env.GITHUB_OUTPUT, `has-merges-multiarch=${hasMergesMultiarch}\n`);
|
|
fs.appendFileSync(process.env.GITHUB_OUTPUT, `matrix-multiarch=${JSON.stringify({ include: multiarch })}\n`);
|
|
fs.appendFileSync(process.env.GITHUB_OUTPUT, `matrix-darwin=${JSON.stringify({ include: includesDarwin })}\n`);
|
|
fs.appendFileSync(process.env.GITHUB_OUTPUT, `merge-matrix-multiarch=${JSON.stringify(mergeMatrixMultiarch)}\n`);
|
|
emitSinglearchShards(singlearch);
|
|
for (const backend of allBackendPaths.keys()) {
|
|
fs.appendFileSync(process.env.GITHUB_OUTPUT, `${backend}=true\n`);
|
|
}
|
|
}
|
|
|
|
function emitFilteredMatrix(changedFiles, previousMatrix, protoRevisions) {
|
|
console.log("Changed files:", changedFiles);
|
|
|
|
const { filtered, filteredDarwin, changedBackends } = filterMatrix({
|
|
includes,
|
|
includesDarwin,
|
|
changedFiles,
|
|
previousMatrix,
|
|
protoRevisions,
|
|
});
|
|
|
|
console.log("Filtered files:", filtered);
|
|
console.log("Filtered files Darwin:", filteredDarwin);
|
|
|
|
const { multiarch, singlearch } = splitByArch(filtered);
|
|
const hasBackendsMultiarch = multiarch.length > 0 ? 'true' : 'false';
|
|
const hasBackendsDarwin = filteredDarwin.length > 0 ? 'true' : 'false';
|
|
console.log("Has single-arch backends?:", singlearch.length > 0 ? 'true' : 'false');
|
|
console.log("Has multi-arch backends?:", hasBackendsMultiarch);
|
|
console.log("Has Darwin backends?:", hasBackendsDarwin);
|
|
|
|
const mergeMatrixMultiarch = computeMergeMatrix(multiarch);
|
|
const hasMergesMultiarch = mergeMatrixMultiarch.include.length > 0 ? 'true' : 'false';
|
|
|
|
fs.appendFileSync(process.env.GITHUB_OUTPUT, `run-all=false\n`);
|
|
fs.appendFileSync(process.env.GITHUB_OUTPUT, `has-backends-multiarch=${hasBackendsMultiarch}\n`);
|
|
fs.appendFileSync(process.env.GITHUB_OUTPUT, `has-backends-darwin=${hasBackendsDarwin}\n`);
|
|
fs.appendFileSync(process.env.GITHUB_OUTPUT, `has-merges-multiarch=${hasMergesMultiarch}\n`);
|
|
fs.appendFileSync(process.env.GITHUB_OUTPUT, `matrix-multiarch=${JSON.stringify({ include: multiarch })}\n`);
|
|
fs.appendFileSync(process.env.GITHUB_OUTPUT, `matrix-darwin=${JSON.stringify({ include: filteredDarwin })}\n`);
|
|
fs.appendFileSync(process.env.GITHUB_OUTPUT, `merge-matrix-multiarch=${JSON.stringify(mergeMatrixMultiarch)}\n`);
|
|
emitSinglearchShards(singlearch);
|
|
|
|
// Per-backend boolean outputs
|
|
for (const backend of allBackendPaths.keys()) {
|
|
const changed = changedBackends.has(backend);
|
|
fs.appendFileSync(process.env.GITHUB_OUTPUT, `${backend}=${changed ? 'true' : 'false'}\n`);
|
|
}
|
|
}
|
|
|
|
(async () => {
|
|
// Tag pushes and an explicit FORCE_ALL escape hatch always rebuild everything.
|
|
// FORCE_ALL is set from backend.yml whenever github.ref starts with refs/tags/.
|
|
const forceAll = process.env.FORCE_ALL === 'true';
|
|
const isTagPush = typeof event.ref === 'string' && event.ref.startsWith('refs/tags/');
|
|
const isBranchPush = !!event.ref && !event.pull_request && !isTagPush;
|
|
|
|
let changedFiles = null;
|
|
if (event.pull_request) {
|
|
changedFiles = await getChangedFilesForPR(event);
|
|
} else if (isBranchPush && !forceAll) {
|
|
changedFiles = await getChangedFilesForPush(event);
|
|
// null -> fall through to the full matrix (e.g. first push, API truncated,
|
|
// network failure).
|
|
}
|
|
// All other event types (workflow_dispatch, schedule, tag pushes, FORCE_ALL)
|
|
// leave changedFiles === null and run everything.
|
|
|
|
if (changedFiles === null) {
|
|
emitFullMatrix();
|
|
return;
|
|
}
|
|
|
|
const previousMatrix = changedFiles.includes(BACKEND_MATRIX_FILE)
|
|
? await getPreviousMatrix(event)
|
|
: null;
|
|
|
|
const protoRevisions = changedFiles.includes(BACKEND_PROTO_FILE)
|
|
? await getProtoRevisions(event)
|
|
: null;
|
|
|
|
emitFilteredMatrix(changedFiles, previousMatrix, protoRevisions);
|
|
})();
|