mirror of
https://github.com/mudler/LocalAI.git
synced 2026-08-06 21:32:59 -04:00
Compare commits
307 Commits
worktree-f
...
fix/stagin
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f7c88770d3 | ||
|
|
36f20f72f8 | ||
|
|
2a8eb5a04b | ||
|
|
9c8f510021 | ||
|
|
1e0baec2a7 | ||
|
|
7bda73fd66 | ||
|
|
a2c87947a9 | ||
|
|
7c984f5c81 | ||
|
|
f8755997cc | ||
|
|
d0401f9bb4 | ||
|
|
0cdd781c2d | ||
|
|
f01038f479 | ||
|
|
0eb8a1188d | ||
|
|
d7e04dcc32 | ||
|
|
a4bab71f27 | ||
|
|
2f33ad6669 | ||
|
|
1cd7d63c7b | ||
|
|
a784cf669f | ||
|
|
65bdbc4ee3 | ||
|
|
0d9d07d3a5 | ||
|
|
f381844403 | ||
|
|
83a0f16a21 | ||
|
|
6e52d0c2ef | ||
|
|
465d488c90 | ||
|
|
1618c2e445 | ||
|
|
9043cbc786 | ||
|
|
0406741a8c | ||
|
|
b5e4413eab | ||
|
|
e55cc3e2a7 | ||
|
|
9d82c37f98 | ||
|
|
f735cb24c0 | ||
|
|
5c607c09d5 | ||
|
|
2f7b292143 | ||
|
|
864c84f48b | ||
|
|
8cef340659 | ||
|
|
c2704dba5b | ||
|
|
92dc326606 | ||
|
|
217fdd2234 | ||
|
|
0e0221b0f5 | ||
|
|
fb4c61d1c9 | ||
|
|
626ae4d51e | ||
|
|
09b85ee00e | ||
|
|
b19afb192a | ||
|
|
963c637130 | ||
|
|
71e98c13a3 | ||
|
|
10211948b5 | ||
|
|
139470cca0 | ||
|
|
078614c701 | ||
|
|
c1efdbeb9e | ||
|
|
7f72dc3412 | ||
|
|
81c407bc40 | ||
|
|
9c43b2da8f | ||
|
|
27955e0a33 | ||
|
|
036eccc32d | ||
|
|
a15b23b775 | ||
|
|
a4a14c6263 | ||
|
|
00cbfc369b | ||
|
|
cee6780ea7 | ||
|
|
79113c7f90 | ||
|
|
e7520af5d7 | ||
|
|
a9456bbce9 | ||
|
|
24c16c9bb5 | ||
|
|
0389495388 | ||
|
|
2f011094d9 | ||
|
|
bc653c9b09 | ||
|
|
f9a2d9be32 | ||
|
|
f40e07d72e | ||
|
|
911fb754a6 | ||
|
|
2dade4a9f9 | ||
|
|
c0a20d6ab1 | ||
|
|
78775c77d8 | ||
|
|
525af1df1b | ||
|
|
279f5b8a93 | ||
|
|
9edb08ea94 | ||
|
|
2ad3b5088b | ||
|
|
a89d780707 | ||
|
|
55e2726958 | ||
|
|
4be6e22b5f | ||
|
|
bf484c5181 | ||
|
|
40d35c0385 | ||
|
|
d3ea65a112 | ||
|
|
4f592c8734 | ||
|
|
6ccb1130d8 | ||
|
|
3f8806b0b2 | ||
|
|
14c7c04feb | ||
|
|
ec933b837d | ||
|
|
cc26083423 | ||
|
|
8aa8e0fac0 | ||
|
|
e9056399a7 | ||
|
|
3bb0d1cb49 | ||
|
|
0bd7a29f31 | ||
|
|
45b8047736 | ||
|
|
fd0d1b946d | ||
|
|
6dfda9c4b6 | ||
|
|
dffcbd7e5d | ||
|
|
7c542fb979 | ||
|
|
cbf232e5fe | ||
|
|
1f53dff436 | ||
|
|
c1a891662c | ||
|
|
06b4a29387 | ||
|
|
e62221b020 | ||
|
|
dc2cc4da43 | ||
|
|
ab7b58fc85 | ||
|
|
bcdb8debfe | ||
|
|
bbe018c1a0 | ||
|
|
3880812ed6 | ||
|
|
808312b4b9 | ||
|
|
6a985d13ea | ||
|
|
5fe48e4910 | ||
|
|
ff8774327f | ||
|
|
688f904a10 | ||
|
|
8c9b3b2e33 | ||
|
|
e488884b20 | ||
|
|
e062179d4d | ||
|
|
b9d6d49e31 | ||
|
|
a23fcc90c3 | ||
|
|
d19c9875ed | ||
|
|
8cec22c3b7 | ||
|
|
3601174ce0 | ||
|
|
40763d1181 | ||
|
|
afbed9d49b | ||
|
|
bcc41219f7 | ||
|
|
d82c38ee77 | ||
|
|
64124f3fa1 | ||
|
|
88cc80ee3d | ||
|
|
bed5e7417c | ||
|
|
ba1d0f5507 | ||
|
|
2bed6f65ba | ||
|
|
b224c96db6 | ||
|
|
9f14571397 | ||
|
|
a5aa56db81 | ||
|
|
05b8d8aafe | ||
|
|
3c1e583985 | ||
|
|
2609848d80 | ||
|
|
cdb6702ca6 | ||
|
|
0d8bea0158 | ||
|
|
0b0f52bedc | ||
|
|
48b1ab28b7 | ||
|
|
b10e330590 | ||
|
|
4056283aa4 | ||
|
|
b90e1cae73 | ||
|
|
c43ee40eb7 | ||
|
|
659b9f02e0 | ||
|
|
67c14e1b7e | ||
|
|
0e6241a5aa | ||
|
|
b00422e45f | ||
|
|
af8f74cba2 | ||
|
|
cdd9582653 | ||
|
|
fb0f5e4bdd | ||
|
|
5013d53a1c | ||
|
|
e8d8f5b0b8 | ||
|
|
459ffb3054 | ||
|
|
cad07be2fc | ||
|
|
2634b13a5d | ||
|
|
8786eace97 | ||
|
|
a1cefe862d | ||
|
|
50cd897719 | ||
|
|
8ff3c8c466 | ||
|
|
1f9fda7138 | ||
|
|
23a044ee0b | ||
|
|
921a8ffc8b | ||
|
|
a8f1c92a24 | ||
|
|
6084497da1 | ||
|
|
2482f075a2 | ||
|
|
9b4f373bc4 | ||
|
|
185956154a | ||
|
|
d3d5488dc7 | ||
|
|
ae58115ee6 | ||
|
|
c7a9db29a6 | ||
|
|
c46a224b44 | ||
|
|
7e8542ba32 | ||
|
|
3c2d85aae4 | ||
|
|
6ceb2f86a7 | ||
|
|
c5b36639d4 | ||
|
|
94bdc825dc | ||
|
|
294487eb61 | ||
|
|
fa3e139540 | ||
|
|
35024338a6 | ||
|
|
b987f39de8 | ||
|
|
16d028a127 | ||
|
|
70e15679cd | ||
|
|
5569b2de56 | ||
|
|
c9f73f40ff | ||
|
|
4cd3dbe931 | ||
|
|
1a04b670f4 | ||
|
|
e948f27965 | ||
|
|
40dae953f4 | ||
|
|
8671c8adac | ||
|
|
d5d659bb65 | ||
|
|
a0ed395cf9 | ||
|
|
40c29db8c4 | ||
|
|
0aaf7cce76 | ||
|
|
731bf04668 | ||
|
|
d829e818d0 | ||
|
|
0ae84be362 | ||
|
|
d521608e6d | ||
|
|
7dde5a4225 | ||
|
|
cd65a1f645 | ||
|
|
97175f4b5a | ||
|
|
1d5139f0a0 | ||
|
|
8565febe45 | ||
|
|
a3fdfbc0d1 | ||
|
|
2f33cc7bc4 | ||
|
|
22225217e0 | ||
|
|
c1fd12a506 | ||
|
|
d01b2c4f46 | ||
|
|
fb9ff061f1 | ||
|
|
40f847745e | ||
|
|
ba9327b9f8 | ||
|
|
fd467c5b3b | ||
|
|
fa0622604a | ||
|
|
ff5758113b | ||
|
|
29db4ab414 | ||
|
|
a6cf67cc6b | ||
|
|
85f5267ed2 | ||
|
|
ed3b59baf1 | ||
|
|
461ae84732 | ||
|
|
2a4426c5ec | ||
|
|
2348bdc16d | ||
|
|
2ccc67bc7f | ||
|
|
0a6c62bb59 | ||
|
|
1297356e29 | ||
|
|
3f36b1dbed | ||
|
|
783222baf4 | ||
|
|
bd3f2588fd | ||
|
|
40e659974d | ||
|
|
deb43e56c0 | ||
|
|
33869da527 | ||
|
|
8059117c2d | ||
|
|
b0959d4756 | ||
|
|
9e41be4bfb | ||
|
|
38350d363e | ||
|
|
817136c20e | ||
|
|
8396ce1388 | ||
|
|
348f3c87c0 | ||
|
|
13310905a3 | ||
|
|
2cbb3c96b3 | ||
|
|
1152acc167 | ||
|
|
cc8ee62db0 | ||
|
|
bfd6c09d88 | ||
|
|
eb32cd9073 | ||
|
|
80ec22945a | ||
|
|
7a3583b52c | ||
|
|
715d4ed8e5 | ||
|
|
9fcc9c0d43 | ||
|
|
3c67b5b746 | ||
|
|
bea66fd84e | ||
|
|
f7a5dfd5ae | ||
|
|
6bcaf30c14 | ||
|
|
ef15b4bfda | ||
|
|
237bce48e8 | ||
|
|
a4e6e01e4d | ||
|
|
6eea3ef2ac | ||
|
|
ad97bcbbdd | ||
|
|
9d8ff90941 | ||
|
|
29001a88c1 | ||
|
|
b0bfa0852e | ||
|
|
39a93e91cf | ||
|
|
26e0c98967 | ||
|
|
9acca54b25 | ||
|
|
2728e6000e | ||
|
|
006310d746 | ||
|
|
05acdb1778 | ||
|
|
5e68b5700c | ||
|
|
7910018249 | ||
|
|
1a03712a6f | ||
|
|
703ea32de6 | ||
|
|
751db06e35 | ||
|
|
f46c0e9c83 | ||
|
|
0d8adfc59a | ||
|
|
43f2615e19 | ||
|
|
875c539ad5 | ||
|
|
d641ded194 | ||
|
|
40445fff05 | ||
|
|
057dee956a | ||
|
|
4ec39bb776 | ||
|
|
25ecb9f015 | ||
|
|
2be495f9c0 | ||
|
|
02b007a31e | ||
|
|
fd8cebd0b3 | ||
|
|
dd625921ff | ||
|
|
d74f88357e | ||
|
|
dfaec3bd51 | ||
|
|
0e381897b5 | ||
|
|
b1af37257d | ||
|
|
ebefa6dcca | ||
|
|
605348925d | ||
|
|
686ce10b54 | ||
|
|
2cee318fad | ||
|
|
1a4f68ed4a | ||
|
|
28d7397743 | ||
|
|
5d0c43ec6e | ||
|
|
6ab29ec8b9 | ||
|
|
036f950b1b | ||
|
|
5b7b914b4f | ||
|
|
d1cee4c52a | ||
|
|
baaa0fe94f | ||
|
|
c3b5c7c3fa | ||
|
|
bd1ec8f2c2 | ||
|
|
135debf9af | ||
|
|
e8c18ae28e | ||
|
|
c4d302e1ab | ||
|
|
323b57a4bc | ||
|
|
3d2f639213 | ||
|
|
be1ae9338b | ||
|
|
923c47020d | ||
|
|
b7a1dec773 |
@@ -34,7 +34,7 @@ The build matrix is data-only YAML at `.github/backend-matrix.yml` (not inside `
|
||||
|
||||
**Without an entry here no image is ever built or pushed, and the gallery entry in `backend/index.yaml` will point at a tag that does not exist.** The `dockerfile:` field must point at `./backend/Dockerfile.<lang>` matching the language bucket from step 1 (e.g. `Dockerfile.python`, `Dockerfile.golang`, `Dockerfile.rust`). The `tag-suffix` must match the `uri:` in the corresponding `backend/index.yaml` image entry exactly.
|
||||
|
||||
**`scripts/changed-backends.js` registration — REQUIRED for any new dockerfile suffix.** This is the single most common omission, because it has no effect on the PR that adds the backend (when no prior path filter could catch it anyway) — it only breaks the *next* PR that touches your backend's directory, which then gets zero CI jobs and looks broken for unrelated reasons. Edit `scripts/changed-backends.js:inferBackendPath` and add a branch BEFORE the more-generic suffixes:
|
||||
**Path-filter registration — REQUIRED for any new dockerfile suffix.** This is the single most common omission, because it has no effect on the PR that adds the backend (when no prior path filter could catch it anyway) — it only breaks the *next* PR that touches your backend's directory, which then gets zero CI jobs and looks broken for unrelated reasons. Edit `scripts/lib/backend-filter.mjs:inferBackendPath` and add a branch BEFORE the more-generic suffixes:
|
||||
|
||||
```js
|
||||
if (item.dockerfile.endsWith("<your-dockerfile-suffix>")) {
|
||||
@@ -54,7 +54,9 @@ for (const e of m.include.filter(e => e.backend === '<your-backend>')) {
|
||||
}"
|
||||
```
|
||||
|
||||
A quick way to find the right insertion point: `grep -n 'item.dockerfile.endsWith' scripts/changed-backends.js`.
|
||||
A quick way to find the right insertion point: `grep -n 'item.dockerfile.endsWith' scripts/lib/backend-filter.mjs`.
|
||||
|
||||
If your backend consumes a *shared* build input that lives outside its own directory (a new script under `scripts/build/`, a new file copied into every image), add a rule to `SHARED_BUILD_INPUTS` in the same file — the per-backend prefix match cannot see those, and a miss ships your change to no image at all. See `scripts/lib/backend-filter_test.mjs` for the pattern; `make test-ci-scripts` runs it.
|
||||
|
||||
**`bump_deps.yaml` registration — REQUIRED for any backend pinning an upstream commit.** If your backend's Makefile has a `*_VERSION?=<sha>` pin to a third-party repo, the daily auto-bump bot at `.github/workflows/bump_deps.yaml` won't notice it unless you register the backend in its matrix. The bot runs `.github/bump_deps.sh` which `grep`s for `^$VAR?=` in the Makefile you list — so the pin MUST live in the Makefile (not in a separate shell script). The bump for ds4 (#9761) had to walk this back because the original landed the pin in `prepare.sh`, which the bot can't see. Pattern (for `antirez/ds4`):
|
||||
|
||||
@@ -115,7 +117,7 @@ Wiring a backend into `includeDarwin:` is more than the matrix entry:
|
||||
|
||||
1. **`includeDarwin:` entry** — `tag-suffix: "-metal-darwin-arm64-<backend>"`, `build-type: "metal"`, `lang: "go"` for go+ggml backends; omit `build-type` for the bespoke C++ ones (llama-cpp / ds4 / privacy-filter). Match an existing entry of the same shape.
|
||||
2. **`backend/index.yaml`** — add `metal:` to the backend's `capabilities` map (main and `-development`) and concrete `metal-<backend>` / `metal-<backend>-development` image entries pointing at the `-metal-darwin-arm64-<backend>` images.
|
||||
3. **C/C++ backends only** — add an `inferBackendPathDarwin` case in `scripts/changed-backends.js` returning `backend/cpp/<backend>/` (the generic fallthrough assumes `backend/<lang>/`, which is wrong for a C++ source tree driven with `lang: go`), and give `run.sh` a Darwin branch that exports `DYLD_LIBRARY_PATH` instead of `LD_LIBRARY_PATH`. If the build is bespoke (single `grpc-server` + dylib bundling), model it on `scripts/build/ds4-darwin.sh` and add a `backends/<backend>-darwin` make target plus a gated step in `.github/workflows/backend_build_darwin.yml`.
|
||||
3. **C/C++ backends only** — add an `inferBackendPathDarwin` case in `scripts/lib/backend-filter.mjs` returning `backend/cpp/<backend>/` (the generic fallthrough assumes `backend/<lang>/`, which is wrong for a C++ source tree driven with `lang: go`), and give `run.sh` a Darwin branch that exports `DYLD_LIBRARY_PATH` instead of `LD_LIBRARY_PATH`. If the build is bespoke (single `grpc-server` + dylib bundling), model it on `scripts/build/ds4-darwin.sh` and add a `backends/<backend>-darwin` make target plus a gated step in `.github/workflows/backend_build_darwin.yml`.
|
||||
4. **C++ proto gotcha** — if the backend compiles the generated gRPC/protobuf in a separate CMake target (e.g. `hw_grpc_proto`), that target must link `protobuf::libprotobuf` + `gRPC::grpc++` so the Homebrew include dirs propagate; otherwise macOS fails with `google/protobuf/runtime_version.h not found` (Linux hides this because apt headers sit in `/usr/include`).
|
||||
|
||||
The CI path filter only builds a backend on a PR when a file under its directory changes, so a darwin-only YAML edit builds nothing — touch a file under `backend/<lang>/<backend>/` (a one-line comment is enough) in the same PR.
|
||||
@@ -216,6 +218,69 @@ docker-build-backends: ... docker-build-<backend-name>
|
||||
- If the backend is in `backend/python/<backend-name>/` but uses `.` as context in the workflow file, use `.` context
|
||||
- Check similar backends to determine the correct context
|
||||
|
||||
## Engine preference for gallery model variants
|
||||
|
||||
A gallery entry can declare `variants`, alternative builds of the same weights,
|
||||
and LocalAI picks one per host: it drops builds whose backend cannot run here or
|
||||
that do not fit memory, then ranks the survivors by **engine preference
|
||||
first, serving feature second, size third** (`SelectVariant` in
|
||||
`core/gallery/resolve_variant.go`).
|
||||
|
||||
Ask whether your backend should outrank another one on some hardware. If it
|
||||
should, add it to `engineNamePreferenceRules` in `pkg/system/capabilities.go`,
|
||||
best engine first for that capability:
|
||||
|
||||
```go
|
||||
{Nvidia, []string{engineVLLM, engineSGLang, engineLlamaCpp}},
|
||||
+ {Nvidia, []string{engineVLLM, engineSGLang, engineMyEngine, engineLlamaCpp}},
|
||||
```
|
||||
|
||||
That is the ENGINE NAME table, matched as a substring of a gallery entry's
|
||||
`backend:` value. Two sibling tables in the same file speak different
|
||||
vocabularies and are matched against different things:
|
||||
|
||||
| Table | Vocabulary | Matched against | Consumer |
|
||||
|-------|-----------|-----------------|----------|
|
||||
| `backendBuildTagPreferenceRules` | build tags (`cuda`, `rocm`, `metal`) | installed build directory names, as a substring | alias resolution in `ListSystemBackends` |
|
||||
| `engineNamePreferenceRules` | engine names (`vllm`, `llama-cpp`, `mlx`) | a gallery entry's `backend:`, as a substring | gallery variant ranking |
|
||||
| `servingFeaturePreferenceTokens` | serving features (`dflash`, `mtp`) | a gallery entry's `tags:`, compared whole and case-insensitively, and nothing else | gallery variant ranking, one rank below the engine |
|
||||
|
||||
**Putting a token in the wrong table matches nothing and does not error**: every
|
||||
candidate scores equal and the next sort key decides, so the preference silently
|
||||
stops existing. The block comment above all three tables spells the contract out.
|
||||
|
||||
The serving feature table is the odd one: it is not keyed by capability, because
|
||||
no hardware prefers a plain build over an equivalent faster build of the same
|
||||
weights. It reads a declared tag and nothing else. The entry name was the
|
||||
original signal and is gone: a naming convention is not a contract, and names
|
||||
are author-supplied free text where a short marker like `mtp` turns up inside
|
||||
unrelated words or on weights whose entry enables nothing.
|
||||
`overrides.options` was rejected for the mirror-image reason: `spec_type:` is
|
||||
llama.cpp's config vocabulary, whereas a cross-backend ranking decision must
|
||||
work the same for `ds4`'s `mtp_path:` and `sglang`'s `speculative_algorithm:`.
|
||||
|
||||
**If your backend can serve the same weights faster** (speculative decoding,
|
||||
multi-token prediction), say so in the docs for its gallery entries so curators
|
||||
tag them: the tagging rule and the per-backend evidence table live in
|
||||
[adding-gallery-models.md](adding-gallery-models.md). A backend never needs to
|
||||
appear in the token table itself; it ranks builds, not engines.
|
||||
|
||||
Leaving your backend out is a valid choice when no ordering can be justified for
|
||||
it. It then ranks below every known engine and selection falls back to size,
|
||||
which is the behaviour that predates preference.
|
||||
|
||||
**Leaving a whole capability out is not.** A missing row gives that host an
|
||||
empty preference list, so size alone decides among everything that survives the
|
||||
filters, and the filter will not save you: `IsBackendCompatible` derives hardware
|
||||
support from the engine NAME, so `vllm` and `sglang` carry no darwin, cuda, rocm
|
||||
or sycl token and are never dropped on a host with no GPU. That is why `default`
|
||||
(no usable accelerator, including a GPU under the 4 GiB VRAM floor) and
|
||||
`darwin-x86` both have rows putting `llama-cpp` first. Every capability
|
||||
`getSystemCapabilities()` can return needs a row unless every engine really is
|
||||
equally at home there. When you add one, enumerate the engines you are demoting
|
||||
rather than relying on them falling through unmatched: unmatched engines all tie
|
||||
with each other, so size decides among them.
|
||||
|
||||
## Documenting the backend (README + docs)
|
||||
|
||||
A backend is not "added" until it is discoverable. Update the user-facing docs:
|
||||
@@ -243,7 +308,7 @@ After adding a new backend, verify:
|
||||
|
||||
- [ ] Backend directory structure is complete with all necessary files
|
||||
- [ ] Build configurations added to `.github/backend-matrix.yml` for all desired platforms (per-arch entries with `platform-tag` for multi-arch; `builder-base-image` for llama-cpp / ik-llama-cpp / turboquant)
|
||||
- [ ] **OS coverage considered**: added to `includeDarwin:` (macOS/Apple Silicon) if the backend can build there — with the `backend/index.yaml` `metal:` capability + `metal-<backend>` image entries, a `run.sh` Darwin/DYLD branch and `inferBackendPathDarwin` case for C++ backends — or the PR explains why an OS is unsupported. Do not ship Linux-only by default.
|
||||
- [ ] **OS coverage considered**: added to `includeDarwin:` (macOS/Apple Silicon) if the backend can build there — with the `backend/index.yaml` `metal:` capability + `metal-<backend>` image entries, a `run.sh` Darwin/DYLD branch and `inferBackendPathDarwin` case (in `scripts/lib/backend-filter.mjs`) for C++ backends — or the PR explains why an OS is unsupported. Do not ship Linux-only by default.
|
||||
- [ ] Meta definition added to `backend/index.yaml` in the `## metas` section
|
||||
- [ ] Image entries added to `backend/index.yaml` for all build variants (latest + development)
|
||||
- [ ] Tag suffixes match between workflow file and index.yaml
|
||||
@@ -251,6 +316,8 @@ After adding a new backend, verify:
|
||||
- [ ] No YAML syntax errors (check with linter)
|
||||
- [ ] No Makefile syntax errors (check with linter)
|
||||
- [ ] Follows the same pattern as similar backends (e.g., if it's a transcription backend, follow `faster-whisper` pattern)
|
||||
- [ ] **`Load` validates its input and refuses models it can't serve.** When a model config has no explicit `backend:`, the model loader greedily probes *every* installed backend with the model's name and binds to the first `Load` that succeeds — an accept-anything `Load` will capture arbitrary LLMs (issue #9287). Backends that load a real artefact get this for free (the load fails); backends with no artefact must gate on the name: `opus` accepts only its own name (or none), `local-store` requires the `store.NamespacePrefix` namespace marker sent by `core/backend/stores.go`.
|
||||
- [ ] **Gallery variant ranking considered**: if this backend should be preferred over another on some hardware, it is listed in `engineNamePreferenceRules` (NOT `backendBuildTagPreferenceRules`, NOT `servingFeaturePreferenceTokens`) in `pkg/system/capabilities.go`. A missing entry silently ranks it last and lets the next sort key decide.
|
||||
- [ ] Documented: added to the category list in `docs/content/features/backends.md` (and any new endpoint/realtime capability documented under `docs/content/`)
|
||||
- [ ] If it is an in-house native C/C++/GGML engine, added to the maintained-engines table in the top-level `README.md`
|
||||
|
||||
|
||||
@@ -91,6 +91,108 @@ To add a variant (e.g., different quantization), use YAML merge:
|
||||
uri: huggingface://<gguf-org>/<gguf-repo>/<filename>-Q8_0.gguf
|
||||
```
|
||||
|
||||
## Offering several builds of one model (`variants`)
|
||||
|
||||
When the same model is published in more than one quantization, or is also
|
||||
servable by another engine, add each build as its own ordinary gallery entry and
|
||||
then point one of them at the others with `variants`:
|
||||
|
||||
```yaml
|
||||
- !!merge <<: *chatml
|
||||
name: "nanbeige4.1-3b-q4"
|
||||
# ... the usual urls / overrides / files for the Q4 build ...
|
||||
variants:
|
||||
- model: nanbeige4.1-3b-q8
|
||||
```
|
||||
|
||||
Rules:
|
||||
|
||||
- The declaring entry is a **complete, normal entry**. It keeps its own
|
||||
`files`/`overrides` and stays installable on every host and by every older
|
||||
LocalAI release, which simply ignore `variants`.
|
||||
- A variant references another gallery entry **by name**. That entry must exist
|
||||
and must not declare `variants` of its own.
|
||||
- **A referenced entry keeps its own gallery row by default.** It is hidden only
|
||||
in the collapsed listing (`collapse_variants=true`, which the web UI requests
|
||||
by default), where the declaring entry stands in for it. Searching there still
|
||||
matches the referenced entry and answers with the entry declaring it, so
|
||||
referencing an entry never makes it unfindable; turning the collapse off
|
||||
returns it under its own name.
|
||||
- **Order carries no meaning.** Do not try to encode a preference; write the
|
||||
list in whatever order reads best.
|
||||
- **A variant may be smaller than the declaring entry.** Offering a downgrade
|
||||
for small hosts is a normal shape: the declaring entry's own build competes
|
||||
like every other candidate, so a large host keeps the large build.
|
||||
- **Do not describe hardware.** At install time LocalAI drops variants whose
|
||||
backend cannot run on the host, then drops those that do not fit available
|
||||
memory. The declaring entry's own build is exempt from both filters, so
|
||||
selection always terminates on something installable. Sizes are measured live
|
||||
from the weights and cached, so nothing has to be written down.
|
||||
- **Engine preference outranks size.** Among the builds that survive the
|
||||
filters, the host's preferred engine wins first and only then does the larger
|
||||
footprint win. On NVIDIA a vLLM build beats a larger llama.cpp one; on Apple
|
||||
silicon an MLX build beats a larger GGUF one; on a host with no preference for
|
||||
either engine the larger build wins, since a bigger footprint is a higher
|
||||
quality quantization of the same weights. Predict what a user gets by asking
|
||||
which engine the host prefers before asking which build is biggest. The
|
||||
per-capability order lives in `engineNamePreferenceRules`
|
||||
(`pkg/system/capabilities.go`); see
|
||||
[adding-backends.md](adding-backends.md) for how a backend gets into it.
|
||||
- **Serving feature preference sits between engine and size.** Among builds on
|
||||
an equally preferred engine, one that speculates or predicts several tokens
|
||||
per step beats the plain build of the same weights, because it answers faster
|
||||
for the same output: a `dflash` build beats an `mtp` one, and either beats a
|
||||
plain build. The order lives in `servingFeaturePreferenceTokens`
|
||||
(`pkg/system/capabilities.go`) and is matched against the entry's `tags:` and
|
||||
**nothing else**: not the entry name, not `overrides.options`. See
|
||||
[the tagging rule](#the-dflash--mtp-tagging-rule) below. Engine deliberately
|
||||
outranks it: a serving feature makes the right engine faster, it does not make
|
||||
a wrong engine right. Fit still outranks both, so a drafter pairing (strictly
|
||||
larger than the plain build, since it ships a drafter alongside it) is dropped
|
||||
on a host too small for it before this order is ever consulted.
|
||||
- A variant is nothing but a name; there is no per-variant memory field. When
|
||||
the measured size for a build is wrong, correct it on the referenced entry by
|
||||
setting that entry's own `size:` (e.g. `size: "20GiB"`). The estimator prefers
|
||||
a declared size over its own guesswork, so the fix applies everywhere the size
|
||||
is shown or compared rather than only to variant selection.
|
||||
|
||||
Users can override the automatic choice with `variant` on `POST /models/apply`,
|
||||
`local-ai models install --variant`, or the `install_model` MCP tool. See
|
||||
`docs/content/features/model-gallery.md`.
|
||||
|
||||
The gallery lint specs live in `core/gallery`, so run that suite after adding a
|
||||
`variants` list.
|
||||
|
||||
### The `dflash` / `mtp` tagging rule
|
||||
|
||||
**Tag an entry `dflash` or `mtp` when the entry actually configures that
|
||||
feature. Variant ranking reads the tag and nothing else.**
|
||||
|
||||
Decide by looking at what the entry configures, in whatever vocabulary its
|
||||
backend uses:
|
||||
|
||||
| Backend | Configures the feature when it declares |
|
||||
|---------|------------------------------------------|
|
||||
| `llama-cpp` | `overrides.options` contains `spec_type:draft-dflash` or `spec_type:draft-mtp` |
|
||||
| `ds4` | `overrides.options` contains `mtp_path:` / `mtp_draft:` |
|
||||
| `sglang` | the referenced `gallery/*.yaml` sets `speculative_algorithm:` |
|
||||
|
||||
That check is curation-time only. `spec_type` is llama.cpp's config vocabulary,
|
||||
and a cross-backend ranking decision must not depend on one backend's option
|
||||
syntax, which is precisely why the ranker reads the tag instead of the options.
|
||||
|
||||
Two mistakes the rule exists to prevent:
|
||||
|
||||
- **Weights that carry the heads are not an entry that enables them.** The
|
||||
NVFP4 GGUF entries ship MTP-bearing weights but set only `use_jinja:true`, so
|
||||
they enable no speculative decoding and must NOT be tagged. Tagging them wins
|
||||
them the feature axis without being any faster.
|
||||
- **A name is not a declaration.** An entry whose name spells `-mtp` while
|
||||
configuring nothing gets no tag, and an entry that configures the feature is
|
||||
tagged even when its name says nothing (`hy3`, `glm-5.2`). Ranking never reads
|
||||
the name, so an untagged build that does enable the feature is simply ranked
|
||||
as plain rather than promoted on a marker nobody meant.
|
||||
|
||||
## Available template configs
|
||||
|
||||
Look at existing `.yaml` files in `gallery/` to find the right prompt template for your model architecture:
|
||||
|
||||
@@ -114,6 +114,24 @@ Both `backend.yml` (push) and `backend_pr.yml` (PR) generate their matrix dynami
|
||||
- **Tag pushes**: `FORCE_ALL=true` is set from the workflow side (`startsWith(github.ref, 'refs/tags/')`) — releases rebuild every backend regardless of diff.
|
||||
- **Schedule / `workflow_dispatch`**: no `event.before`, falls through to "run everything" automatically.
|
||||
|
||||
### Shared build inputs
|
||||
|
||||
The per-backend prefix match only sees files under a backend's own directory, so a change to shared build infrastructure would rebuild *nothing* — an empty matrix, every job green, and the change reaching no image. That silently un-shipped PR #10946 (a partial-cuDNN packaging fix in `scripts/build/package-gpu-libs.sh`), which merged 1h48m after the weekly cron and so sat unbuilt for a week.
|
||||
|
||||
`SHARED_BUILD_INPUTS` in `scripts/lib/backend-filter.mjs` closes that hole. Each rule maps a shared path to the narrowest set of matrix entries it can honestly invalidate, since a full matrix is 417 Linux + 56 Darwin builds:
|
||||
|
||||
| Changed path | Rebuilds |
|
||||
|---|---|
|
||||
| `backend/backend.proto` | everything (all languages compile or copy it) |
|
||||
| `backend/Dockerfile.<x>` | the Linux entries whose `dockerfile:` names it |
|
||||
| `backend/python/common/` | Python, Linux + Darwin |
|
||||
| `scripts/build/package-gpu-libs.sh` | Python, Linux only |
|
||||
| `scripts/build/<lang>-darwin.sh` | the Darwin entries that build target routes to |
|
||||
| `.github/workflows/backend_build[_darwin].yml` | everything on that OS |
|
||||
| anything else under `scripts/build/` (except `*_test.sh`) | everything — conservative default for unclassified packaging inputs |
|
||||
|
||||
Deliberately excluded: `backend/index.yaml` (gallery metadata, never enters an image), `.github/backend-matrix.yml` (adding a backend would rebuild all of them), `backend/Dockerfile.base-grpc-builder` (owned by `base-images.yml`), and the root `Makefile` (touched in ~11% of commits, and its backend-relevant edits arrive alongside the backend directory anyway). `make test-ci-scripts` pins all of this.
|
||||
|
||||
The Sunday 06:00 UTC cron on `backend.yml` exists specifically because path filtering can leave Python backends frozen on stale wheels. `DEPS_REFRESH` (below) only fires when the build actually runs, so an untouched Python backend would never re-resolve its unpinned deps. The weekly cron is the safety net.
|
||||
|
||||
## The `DEPS_REFRESH` cache-buster (Python backends)
|
||||
|
||||
@@ -65,6 +65,7 @@ This is enforced by `forbidigo` (see `.golangci.yml`): `http.DefaultClient` and
|
||||
|
||||
The project documentation is located in `docs/content`. When adding new features or changing existing functionality, it is crucial to update the documentation to reflect these changes. This helps users understand how to use the new capabilities and ensures the documentation stays relevant.
|
||||
|
||||
- **Docs-with-code rule**: When you change user-facing behavior (API endpoints, CLI flags, config keys, or features), update the corresponding page under `docs/content/` in the SAME change, not as a follow-up. A user-facing change without a matching docs update is incomplete. The PR template carries a checklist item for this.
|
||||
- **Feature Documentation**: If you add a new feature (like a new backend or API endpoint), create a new markdown file in `docs/content/features/` explaining what it is, how to configure it, and how to use it.
|
||||
- **Configuration**: If you modify configuration options, update the relevant sections in `docs/content/`.
|
||||
- **Examples**: providing concrete examples (like YAML configuration blocks) is highly encouraged to help users get started quickly.
|
||||
|
||||
@@ -1,143 +0,0 @@
|
||||
# llama-cpp-localai-paged Backend (paged attention + Blackwell NVFP4 decode)
|
||||
|
||||
`llama-cpp-localai-paged` is LocalAI's **CUDA-only** paged-attention variant of the
|
||||
llama.cpp backend. It targets high-concurrency decode for the Qwen3.6 hybrid
|
||||
gated-DeltaNet (SSM) models on Blackwell (GB10 / DGX Spark). It reuses the stock
|
||||
`llama-cpp` backend's sources and applies a vendored patch series on top at build
|
||||
time. It is **not** a fork: a source-only `*.patch` stack plus one canonical doc.
|
||||
|
||||
**Canonical reference:** `backend/cpp/llama-cpp-localai-paged/README.md`
|
||||
(architecture, the patch series 0001-0030, benchmarks, dev notes, generality,
|
||||
pin/canary policy). Read it for any technical detail; this guide is the maintenance
|
||||
how-to.
|
||||
|
||||
## Where things live
|
||||
|
||||
- `backend/cpp/llama-cpp-localai-paged/Makefile` - the thin wrapper. It copies the
|
||||
stock `backend/cpp/llama-cpp/` build infra into a build dir, clones llama.cpp at
|
||||
this backend's **own** pin (`LLAMA_VERSION`), applies the paged series via the
|
||||
`apply-paged-patches` define (strict `git apply`), then builds `grpc-server`.
|
||||
- `backend/cpp/llama-cpp-localai-paged/patches/paged/` - the source-only `.patch`
|
||||
series (0001-0030), nothing else.
|
||||
- `backend/cpp/llama-cpp-localai-paged/README.md` - the canonical doc. The
|
||||
operational docs (`PAGED_BITEXACT_NOTE.md`, `UPSTREAM_LAYER2_SCOPE.md`) and
|
||||
dev artifacts live in
|
||||
`backend/cpp/llama-cpp-localai-paged/docs/`.
|
||||
- `backend/Dockerfile.llama-cpp-localai-paged`, `.docker/llama-cpp-localai-paged-compile.sh`
|
||||
- the CUDA build entry points.
|
||||
- `backend/cpp/llama-cpp/` - the **stock** backend, pure upstream. It carries no
|
||||
paged patches.
|
||||
|
||||
## Invariants (do not break these)
|
||||
|
||||
- **Stock stays pure.** The paged patches live ONLY in this backend. Never add a
|
||||
`patches/paged/` dir or `LLAMA_PAGED` logic to `backend/cpp/llama-cpp/`.
|
||||
- **CUDA-only.** Ship cublas/cuda targets only. Off-CUDA the fusions are gated off
|
||||
(patch 0030) and NVFP4 falls back to dequant, so the backend is neutral-to-
|
||||
slightly-negative there - non-CUDA users use the stock `llama-cpp`. Do not add
|
||||
cpu/vulkan/sycl/metal rows for this backend in `.github/backend-matrix.yml`.
|
||||
(Those builds also fail to link `grpc-server` on darwin/arm64 against upstream
|
||||
`stream_*` server symbols - another reason it is CUDA-only.)
|
||||
- **Source-only patches.** A `.patch` may touch only llama.cpp source - never a
|
||||
dev doc or `*.md`. Strict `git apply` on a clean checkout must reach exit 0. (A
|
||||
stray `SSM_DECODE_FIX_RESULTS.md` hunk in patch 0019 once broke the CI build.)
|
||||
- **Bit-exact by default.** Every shipped patch is byte-identical to the f32
|
||||
baseline. (The one opt-in precision trade, `ssm_bf16_tau` / patch 0026, was
|
||||
DROPPED: it went flat once the decode fusions landed - forcing all gated-DeltaNet
|
||||
heads to bf16 gave 780.6 vs 780.0 t/s, zero benefit - so the series is now
|
||||
bit-exact end to end. Do not reintroduce a per-head SSM-precision lever; see the
|
||||
rejected-levers note in the backend README section 5.)
|
||||
|
||||
## Fork-first workflow (MANDATORY)
|
||||
|
||||
The fork **`mudler/llama.cpp` branch `localai-paged`** is the CANONICAL source
|
||||
of truth for ALL paged-backend kernel and patch work. The vendored
|
||||
`patches/paged/*.patch` series is a **derivative**: the fork is the source, the
|
||||
series is a generated mirror of it.
|
||||
|
||||
**Always update the fork FIRST, in this exact order:**
|
||||
|
||||
1. **Commit the change on the `localai-paged` branch and push it.** Every
|
||||
kernel or patch change lands as a fork commit first.
|
||||
2. **Then regenerate the LocalAI series from the fork** via `git format-patch`
|
||||
(one patch per fork commit, source-only) into
|
||||
`backend/cpp/llama-cpp-localai-paged/patches/paged/`, so the series stays a
|
||||
**1:1, drift-free mirror** of the branch.
|
||||
|
||||
Hard rules, no exceptions:
|
||||
|
||||
- **NEVER edit the `patches/paged/*.patch` files directly.** They are generated
|
||||
output, not source.
|
||||
- **NEVER add a patch to the series that has no corresponding fork-branch
|
||||
commit.** Every `.patch` must be the `git format-patch` of a real commit on
|
||||
`localai-paged`.
|
||||
- The fork branch is **where the build and the per-path bit-exact md5 gate
|
||||
actually run**, so it is the **only** place a change is truly validated. A
|
||||
patch living only in the LocalAI series has never been built or gated.
|
||||
|
||||
Verify the mirror by tree hash: applying the full on-disk series on the pin
|
||||
must reproduce the fork branch tree byte-for-byte. (The patch maintenance
|
||||
detail is in `backend/cpp/llama-cpp-localai-paged/docs/PATCH_MAINTENANCE.md`;
|
||||
the hard-gate is section 2.5 of `docs/PARITY_HANDOFF.md`.)
|
||||
|
||||
## Maintaining the pin against new llama.cpp
|
||||
|
||||
The pin (`LLAMA_VERSION` in the wrapper Makefile) is advanced ONLY by the manual
|
||||
pin-sync. It is deliberately **excluded from the nightly auto-bumper**
|
||||
(`bump_deps.yaml`): a naive bump would shift the tree out from under the patches
|
||||
and break `git apply` at build time.
|
||||
|
||||
1. **The canary tells you when to sync.** `.github/workflows/llama-cpp-paged-canary.yml`
|
||||
runs weekly: it applies + builds the series against the latest upstream tip and
|
||||
goes **red** when upstream drifts past the patches. Canary red -> run a pin-sync.
|
||||
2. **The pin-sync** (recorded in the README section 7 and git history): rebase the series onto the new
|
||||
tip (resolve conflicts; re-export **source-only** with a pathspec like
|
||||
`-- src/ ggml/ common/ include/ tools/ tests/ cmake/`), rebuild on a CUDA box,
|
||||
pass the bit-exact gate on **every** path + `test-backend-ops`, **and confirm
|
||||
the full grpc-server build/link is green on CI**, then bump `LLAMA_VERSION`.
|
||||
|
||||
**Hard constraint: keep the pin == the stock `llama-cpp` pin.** `grpc-server.cpp`
|
||||
is shared with the stock backend and tracks the stock pin. A paged pin that
|
||||
diverges PAST an upstream server-API refactor breaks the grpc-server LINK even
|
||||
when the patches are byte-for-byte bit-exact - the bit-exact gate alone does NOT
|
||||
catch it. The `c299a92c` bump did exactly this (patches applied + greedy-md5
|
||||
bit-exact, but `grpc-server.cpp` failed to link with undefined `stream_*` server
|
||||
helpers the refactor pulled into its headers), so it was reverted to `9d5d882d`.
|
||||
A pin bump is shippable only once the full CI grpc-server build is green, which in
|
||||
practice means moving in lockstep with the stock pin (or vendoring a
|
||||
pin-matched grpc-server.cpp, which we deliberately do not, to keep stock pure).
|
||||
|
||||
## The bit-exact gate (run for every change)
|
||||
|
||||
- greedy md5: `llama-completion -m MODEL -ngl 99 -fa on -p "The capital of France is" -n 48 --temp 0 --seed 1 </dev/null | md5sum`,
|
||||
paged paths prefixed `LLAMA_KV_PAGED=1` (+ `LLAMA_MOE_FORCE_GRAPHS=1` for paged
|
||||
MoE). Must match the recorded baseline. Redirect stdin from `/dev/null` or
|
||||
`llama-completion` hangs in conversation mode.
|
||||
- `test-backend-ops` (CUDA0 vs CPU oracle) for every touched op (`SSM_CONV*`,
|
||||
`GATED_DELTA_NET`, `MUL_MAT`, `MUL_MAT_ID`).
|
||||
- **The gate is per-path.** The paged-MoE md5 differs from the non-paged md5 - a
|
||||
benign, KL-validated FP-accumulation-order difference (see `docs/PAGED_BITEXACT_NOTE.md`).
|
||||
Compare a paged-MoE change to the **paged** reference, not the non-paged one.
|
||||
|
||||
## Encapsulating your work
|
||||
|
||||
- When you change a kernel, follow the **Fork-first workflow** above: commit and
|
||||
push on the `localai-paged` branch first, then regenerate the `.patch`
|
||||
(source-only) from the fork so this worktree mirrors the branch byte-for-byte.
|
||||
Commit with sign-off.
|
||||
- New optimization -> next patch number (gaps 0005/0027 are intentional). Update
|
||||
the README's patch table and dev notes - keep the README the single doc; do not
|
||||
scatter `*_RESULTS.md` files.
|
||||
- Record rejected/flat levers in the README too (they stop the next person from
|
||||
re-running dead ends).
|
||||
|
||||
## Follow-ups (Metal / SYCL / Vulkan)
|
||||
|
||||
The decode fusions are implemented for **CUDA + CPU only**. The base
|
||||
gated-DeltaNet + SSM_CONV ops already exist upstream on Metal, SYCL, and Vulkan,
|
||||
so the models **run** there via the non-fused path - what is missing is the
|
||||
fusion speedup. Porting it (strictly mirroring the CUDA kernels, since we have no
|
||||
Metal/SYCL/Vulkan hardware to test on here) is scoped in `docs/UPSTREAM_LAYER2_SCOPE.md`
|
||||
(recommended order: Metal, then SYCL, then Vulkan; ops-first upstream PR, then one
|
||||
PR per backend, each gated by `test-backend-ops` on the target hardware). The
|
||||
methodology for that work is in [.agents/vllm-parity-methodology.md](vllm-parity-methodology.md).
|
||||
@@ -1,101 +0,0 @@
|
||||
# Methodology: Closing the vLLM Decode-Throughput Gap in llama.cpp
|
||||
|
||||
This is the playbook that took the paged backend
|
||||
([.agents/llama-cpp-localai-paged-backend.md](llama-cpp-localai-paged-backend.md))
|
||||
from ~38% of vLLM decode to **parity-to-ahead on dense** (and a proven, honest
|
||||
ceiling on MoE) on GB10. Use it for any "make llama.cpp match or beat engine X on
|
||||
accelerator Y" effort. The *levers* are model- and hardware-specific; the
|
||||
*discipline* below is not. The worked example, with all numbers, is the paged
|
||||
backend README.
|
||||
|
||||
## The core loop
|
||||
|
||||
1. **Establish a bit-exact baseline and gate FIRST.** Record the greedy md5 (per
|
||||
path) and an f32 reference. Every optimization must stay byte-identical to it -
|
||||
or ship as an explicit, default-off precision opt-in. This is what lets you
|
||||
optimize aggressively without silently regressing quality. Gate two ways:
|
||||
greedy md5, and `test-backend-ops` against the CPU oracle.
|
||||
|
||||
2. **Profile - do not assume.** nsys the steady-state decode step, broken down per
|
||||
*kernel* AND per *memcpy*. Find the dominant cost. "It's the GEMM" was wrong
|
||||
here: on hybrid gated-DeltaNet models the bottleneck was the recurrent-state
|
||||
**plumbing** (state memcpy + gathers, ~67% of the step), not the weight GEMM.
|
||||
Also sanity-check GPU-busy %: an early "low utilization" reading was a profiling
|
||||
window artifact (decode was 96-99% GPU-busy), not real idle.
|
||||
|
||||
3. **Ground-truth BOTH engines.** Decompose *your* decode step AND the
|
||||
competitor's, side by side, per bucket, and compute the per-bucket delta. This
|
||||
tells you WHERE the gap actually is - not where you would guess. It overturned
|
||||
premises here: e.g. vLLM does NOT run the GDN/attn projections as NVFP4 (it
|
||||
keeps them bf16, same as us); the MoE expert GEMM was a llama *win*, not the gap.
|
||||
|
||||
4. **Per-lever discipline.** For each candidate: implement -> bit-exact gate ->
|
||||
same-harness A/B bench. Use a runtime env-toggle (flag off vs on) ONLY for
|
||||
levers that are actually runtime-gated; a lever **compiled into** the binary
|
||||
(e.g. the SSM decode fusions here) is NOT isolated by a runtime flag, so measure
|
||||
it build-vs-build. The full-patchset "stock" baseline likewise needs a
|
||||
**separately-built unpatched binary at the same pin** - toggling the runtime
|
||||
flag on the patched binary does not reproduce stock (it measures only the gated
|
||||
part; here that was ~neutral, which is exactly how this gotcha hides). Bank only
|
||||
what lifts AND gates. **Record every rejected or flat lever with the reason** -
|
||||
over time this is the most valuable part: it stops the next person re-running
|
||||
dead ends.
|
||||
|
||||
5. **Name the structural floor.** Prove the bit-exact ceiling exhaustively (every
|
||||
lever measured, not assumed). What remains is physical - the memory-bandwidth
|
||||
floor, the irreducible serial-SSM host loop (sampling can't start until logits
|
||||
land). Name it; do not claim more than you measured.
|
||||
|
||||
## Hard rules learned
|
||||
|
||||
- **Apples-to-apples, or label it.** Stock-vs-patched on the SAME harness
|
||||
(`llama-batched-bench`) is exact - lead with it. But "stock" must be a
|
||||
separately-built unpatched binary at the SAME pin, NOT the patched binary with
|
||||
the runtime flag off (compiled-in wins survive the toggle). Cross-engine "% of vLLM"
|
||||
(batched-bench vs vLLM server+client) is *indicative*; always caveat the harness
|
||||
and config (context length alone shifted the MoE figure 76% <-> 86%).
|
||||
- **Re-measure a "win" after later levers land - it may evaporate.** bf16 SSM
|
||||
state (the `ssm_bf16_tau` lever) benched +12% early and failed the f32 KL gate
|
||||
(vLLM keeps f32 too), so it was kept default-off opt-in. Once the decode fusions
|
||||
(recurrent-state gather-fusion + block-table cache) landed, a clean re-measure
|
||||
forcing ALL gated-DeltaNet heads to bf16 (`tau=100000`) went **flat** - 780.6 vs
|
||||
780.0 t/s. The "+12%" was subsumed by the fusions: the lever bought nothing, so
|
||||
it was **dropped** (precision trade + bug surface + extra CUDA template-instantiation
|
||||
compile cost, zero benefit). A win measured before the rest of the series is not a
|
||||
win after it.
|
||||
- **Reject the obvious-but-wrong, with evidence.** A faster kernel that is off the
|
||||
critical path benches FLAT (the freed time becomes idle). Quantizing the bf16
|
||||
projections to NVFP4 cost ~6% PPL - and vLLM keeps them bf16 for the same reason.
|
||||
Always measure before believing; a plausible mechanism is not a result.
|
||||
- **The gate can be per-path.** Paged vs non-paged attention legitimately produces
|
||||
different (equivalent) FP-reduction orders; validate the difference is benign
|
||||
(KLD to f32) and then gate each path against its own reference.
|
||||
|
||||
## Orchestration (multi-agent)
|
||||
|
||||
- **One GPU profiler/bencher at a time** (the GPU-contention rule). Parallel
|
||||
design/analysis/read agents are fine; concurrent GPU benches pollute each other's
|
||||
numbers.
|
||||
- **Adversarial verify.** Before banking a finding, spawn skeptics prompted to
|
||||
*refute* it; majority-refute kills it. Prevents plausible-but-wrong results.
|
||||
- **Anti-punt.** Use foreground, blocking ssh loops with short benches and a
|
||||
progress-file checkpoint. Agents that background work and "wait for the monitor
|
||||
event" stall - forbid that pattern.
|
||||
- **GPU coexistence.** On a shared host, stop the user's deployments for a clean
|
||||
benchmark window (with their OK) and ALWAYS restore them (wrap the bench so a
|
||||
failure cannot strand them).
|
||||
|
||||
## What generalizes (and what doesn't)
|
||||
|
||||
The *speedups* may be hardware-specific (here: CUDA/Blackwell - the SSM fusions,
|
||||
NVFP4 FP4-MMA, the occupancy tune), which is why other accelerators did not
|
||||
benefit. But the *findings* often generalize and are worth upstreaming: the
|
||||
"decode is plumbing-bound, not GEMM-bound" insight and the bit-exact, CPU-mirrored
|
||||
fusion ops help any backend running these models. Separate "ship our tuned backend"
|
||||
from "upstream the portable op" - they are different deliverables.
|
||||
|
||||
## The closing record
|
||||
|
||||
Write up the result HONESTLY: the shipped wins, the rejected levers (with reasons),
|
||||
the structural ceiling, and the cross-backend / cross-quant generality. Negative
|
||||
results are as valuable as wins. The paged backend README is the template.
|
||||
@@ -1,5 +1,5 @@
|
||||
#!/usr/bin/env bash
|
||||
# Shared compile logic for backend/Dockerfile.llama-cpp-localai-paged.
|
||||
# Shared compile logic for backend/Dockerfile.bonsai.
|
||||
# Sourced (via bind mount) from both builder-fromsource and builder-prebuilt stages.
|
||||
|
||||
set -euxo pipefail
|
||||
@@ -14,10 +14,10 @@ if [[ -n "${CUDA_DOCKER_ARCH:-}" ]]; then
|
||||
CUDA_ARCH_ESC="${CUDA_DOCKER_ARCH//;/\\;}"
|
||||
export CMAKE_ARGS="${CMAKE_ARGS} -DCMAKE_CUDA_ARCHITECTURES=${CUDA_ARCH_ESC}"
|
||||
echo "CMAKE_ARGS(env) = ${CMAKE_ARGS}"
|
||||
rm -rf /LocalAI/backend/cpp/llama-cpp-localai-paged-*-build
|
||||
rm -rf /LocalAI/backend/cpp/bonsai-*-build
|
||||
fi
|
||||
|
||||
cd /LocalAI/backend/cpp/llama-cpp-localai-paged
|
||||
cd /LocalAI/backend/cpp/bonsai
|
||||
|
||||
if [ -z "${BUILD_TYPE:-}" ]; then
|
||||
# Pure CPU image: one ggml CPU_ALL_VARIANTS build replaces the per-microarch binaries.
|
||||
@@ -26,14 +26,14 @@ if [ -z "${BUILD_TYPE:-}" ]; then
|
||||
apt-get update -qq && apt-get install -y -qq gcc-14 g++-14
|
||||
export CC=gcc-14 CXX=g++-14
|
||||
fi
|
||||
make llama-cpp-localai-paged-cpu-all
|
||||
make bonsai-cpu-all
|
||||
else
|
||||
# GPU build (cublas/hipblas/sycl/vulkan/...): single fallback CPU build, the accelerator
|
||||
# does the compute. Keeps the GPU compile from also building the CPU variant matrix and
|
||||
# avoids the gcc-14 apt step on GPU base images such as nvidia l4t.
|
||||
make llama-cpp-localai-paged-fallback
|
||||
make bonsai-fallback
|
||||
fi
|
||||
make llama-cpp-localai-paged-grpc
|
||||
make llama-cpp-localai-paged-rpc-server
|
||||
make bonsai-grpc
|
||||
make bonsai-rpc-server
|
||||
|
||||
ccache -s || true
|
||||
@@ -7,8 +7,11 @@
|
||||
# Runs only the checks relevant to what's staged:
|
||||
# - Go files -> make lint + make test-coverage-check
|
||||
# - core/http/react-ui -> make test-ui-coverage-check (Playwright e2e + gate)
|
||||
# A commit touching neither is skipped entirely (docs/YAML/etc. can't change
|
||||
# lint findings, Go coverage, or the UI).
|
||||
# - realtime state machines / specs -> make test-realtime-conformance
|
||||
# (respcoord/**, turncoord/**, or formal-verification/** -- a pure .fizz
|
||||
# spec edit must still re-verify the design, detected separately from Go)
|
||||
# A commit touching none of these is skipped entirely (other docs/YAML can't
|
||||
# change lint findings, Go coverage, the UI, or the realtime conformance gate).
|
||||
#
|
||||
# To bypass for a single commit (e.g. a WIP checkpoint): git commit --no-verify
|
||||
set -eu
|
||||
@@ -20,11 +23,13 @@ staged="$(git diff --cached --name-only --diff-filter=ACMRD)"
|
||||
|
||||
go_changed=0
|
||||
ui_changed=0
|
||||
rt_changed=0
|
||||
if echo "$staged" | grep -qE '\.go$'; then go_changed=1; fi
|
||||
if echo "$staged" | grep -qE '^core/http/react-ui/'; then ui_changed=1; fi
|
||||
if echo "$staged" | grep -qE '^(core/http/endpoints/openai/(coordinator|respcoord|turncoord|conncoord|compactcoord|ttscoord)/|formal-verification/)'; then rt_changed=1; fi
|
||||
|
||||
if [ "$go_changed" -eq 0 ] && [ "$ui_changed" -eq 0 ]; then
|
||||
echo "pre-commit: no Go or React UI changes staged — skipping."
|
||||
if [ "$go_changed" -eq 0 ] && [ "$ui_changed" -eq 0 ] && [ "$rt_changed" -eq 0 ]; then
|
||||
echo "pre-commit: no Go, React UI, or realtime-spec changes staged — skipping."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
@@ -57,4 +62,11 @@ if [ "$ui_changed" -eq 1 ]; then
|
||||
make test-ui-coverage-check
|
||||
fi
|
||||
|
||||
if [ "$rt_changed" -eq 1 ]; then
|
||||
echo "pre-commit ▶ realtime state-machine conformance (make test-realtime-conformance) —"
|
||||
echo " Go transition/rapid tests under -race + FizzBee model check of the"
|
||||
echo " authoritative specs. Fail-closed: needs FizzBee (make install-fizzbee)."
|
||||
make test-realtime-conformance
|
||||
fi
|
||||
|
||||
echo "pre-commit ✓ all relevant checks passed"
|
||||
|
||||
1
.github/PULL_REQUEST_TEMPLATE.md
vendored
1
.github/PULL_REQUEST_TEMPLATE.md
vendored
@@ -7,6 +7,7 @@ This PR fixes #
|
||||
|
||||
**[Signed commits](../CONTRIBUTING.md#signing-off-on-commits-developer-certificate-of-origin)**
|
||||
- [ ] Yes, I signed my commits.
|
||||
- [ ] Documentation updated (docs/content/) for user-facing changes, or not applicable
|
||||
|
||||
<!--
|
||||
Thank you for contributing to LocalAI!
|
||||
|
||||
575
.github/backend-matrix.yml
vendored
575
.github/backend-matrix.yml
vendored
@@ -23,7 +23,7 @@
|
||||
# checklist in .agents/adding-backends.md (includeDarwin entry, the index.yaml
|
||||
# `metal:` capability + `metal-<backend>` image entries, a `run.sh` Darwin/DYLD
|
||||
# branch for C/C++ backends, and the inferBackendPathDarwin case in
|
||||
# scripts/changed-backends.js so the path filter actually builds it).
|
||||
# scripts/lib/backend-filter.mjs so the path filter actually builds it).
|
||||
|
||||
# Linux matrix (consumed by backend-jobs).
|
||||
include:
|
||||
@@ -452,6 +452,22 @@ include:
|
||||
dockerfile: "./backend/Dockerfile.turboquant"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'cublas'
|
||||
cuda-major-version: "12"
|
||||
cuda-minor-version: "8"
|
||||
platforms: 'linux/amd64'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-gpu-nvidia-cuda-12-bonsai'
|
||||
builder-base-image: 'quay.io/go-skynet/ci-cache:base-grpc-cuda-12-amd64'
|
||||
# bigger-runner: same rationale as -gpu-nvidia-cuda-12-llama-cpp above
|
||||
# (observed 6h5m wall-clock on v4.2.1, just past the 6h job timeout).
|
||||
runs-on: 'bigger-runner'
|
||||
base-image: "ubuntu:24.04"
|
||||
skip-drivers: 'false'
|
||||
backend: "bonsai"
|
||||
dockerfile: "./backend/Dockerfile.bonsai"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'cublas'
|
||||
cuda-major-version: "12"
|
||||
cuda-minor-version: "8"
|
||||
@@ -478,6 +494,19 @@ include:
|
||||
dockerfile: "./backend/Dockerfile.python"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'cublas'
|
||||
cuda-major-version: "12"
|
||||
cuda-minor-version: "8"
|
||||
platforms: 'linux/amd64'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-gpu-nvidia-cuda-12-longcat-video'
|
||||
runs-on: 'ubuntu-latest'
|
||||
base-image: "ubuntu:24.04"
|
||||
skip-drivers: 'false'
|
||||
backend: "longcat-video"
|
||||
dockerfile: "./backend/Dockerfile.python"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'cublas'
|
||||
cuda-major-version: "12"
|
||||
cuda-minor-version: "8"
|
||||
@@ -790,6 +819,19 @@ include:
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'cublas'
|
||||
cuda-major-version: "12"
|
||||
cuda-minor-version: "8"
|
||||
platforms: 'linux/amd64'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-gpu-nvidia-cuda-12-moss-transcribe-cpp'
|
||||
runs-on: 'ubuntu-latest'
|
||||
base-image: "ubuntu:24.04"
|
||||
skip-drivers: 'false'
|
||||
backend: "moss-transcribe-cpp"
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'cublas'
|
||||
cuda-major-version: "12"
|
||||
cuda-minor-version: "8"
|
||||
@@ -816,6 +858,19 @@ include:
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'cublas'
|
||||
cuda-major-version: "12"
|
||||
cuda-minor-version: "8"
|
||||
platforms: 'linux/amd64'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-gpu-nvidia-cuda-12-moss-tts-cpp'
|
||||
runs-on: 'ubuntu-latest'
|
||||
base-image: "ubuntu:24.04"
|
||||
skip-drivers: 'false'
|
||||
backend: "moss-tts-cpp"
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'cublas'
|
||||
cuda-major-version: "12"
|
||||
cuda-minor-version: "8"
|
||||
@@ -1056,6 +1111,21 @@ include:
|
||||
dockerfile: "./backend/Dockerfile.turboquant"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'cublas'
|
||||
cuda-major-version: "13"
|
||||
cuda-minor-version: "0"
|
||||
platforms: 'linux/amd64'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-gpu-nvidia-cuda-13-bonsai'
|
||||
builder-base-image: 'quay.io/go-skynet/ci-cache:base-grpc-cuda-13-amd64'
|
||||
# bigger-runner: observed 6h5m wall-clock on v4.2.1 — at the GHA timeout.
|
||||
runs-on: 'bigger-runner'
|
||||
base-image: "ubuntu:24.04"
|
||||
skip-drivers: 'false'
|
||||
backend: "bonsai"
|
||||
dockerfile: "./backend/Dockerfile.bonsai"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'cublas'
|
||||
cuda-major-version: "13"
|
||||
cuda-minor-version: "0"
|
||||
@@ -1084,6 +1154,20 @@ include:
|
||||
backend: "turboquant"
|
||||
dockerfile: "./backend/Dockerfile.turboquant"
|
||||
context: "./"
|
||||
- build-type: 'cublas'
|
||||
cuda-major-version: "13"
|
||||
cuda-minor-version: "0"
|
||||
platforms: 'linux/arm64'
|
||||
skip-drivers: 'false'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-nvidia-l4t-cuda-13-arm64-bonsai'
|
||||
builder-base-image: 'quay.io/go-skynet/ci-cache:base-grpc-cuda-13-arm64'
|
||||
base-image: "ubuntu:24.04"
|
||||
runs-on: 'ubuntu-24.04-arm'
|
||||
ubuntu-version: '2404'
|
||||
backend: "bonsai"
|
||||
dockerfile: "./backend/Dockerfile.bonsai"
|
||||
context: "./"
|
||||
- build-type: 'cublas'
|
||||
cuda-major-version: "13"
|
||||
cuda-minor-version: "0"
|
||||
@@ -1136,6 +1220,19 @@ include:
|
||||
dockerfile: "./backend/Dockerfile.python"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'cublas'
|
||||
cuda-major-version: "13"
|
||||
cuda-minor-version: "0"
|
||||
platforms: 'linux/amd64'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-gpu-nvidia-cuda-13-longcat-video'
|
||||
runs-on: 'ubuntu-latest'
|
||||
base-image: "ubuntu:24.04"
|
||||
skip-drivers: 'false'
|
||||
backend: "longcat-video"
|
||||
dockerfile: "./backend/Dockerfile.python"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'cublas'
|
||||
cuda-major-version: "13"
|
||||
cuda-minor-version: "0"
|
||||
@@ -1344,6 +1441,19 @@ include:
|
||||
backend: "vllm-omni"
|
||||
dockerfile: "./backend/Dockerfile.python"
|
||||
context: "./"
|
||||
- build-type: 'l4t'
|
||||
cuda-major-version: "13"
|
||||
cuda-minor-version: "0"
|
||||
platforms: 'linux/arm64'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-nvidia-l4t-cuda-13-arm64-longcat-video'
|
||||
runs-on: 'ubuntu-24.04-arm'
|
||||
base-image: "ubuntu:24.04"
|
||||
skip-drivers: 'false'
|
||||
ubuntu-version: '2404'
|
||||
backend: "longcat-video"
|
||||
dockerfile: "./backend/Dockerfile.python"
|
||||
context: "./"
|
||||
- build-type: 'l4t'
|
||||
cuda-major-version: "13"
|
||||
cuda-minor-version: "0"
|
||||
@@ -1721,6 +1831,19 @@ include:
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'cublas'
|
||||
cuda-major-version: "13"
|
||||
cuda-minor-version: "0"
|
||||
platforms: 'linux/amd64'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-gpu-nvidia-cuda-13-moss-transcribe-cpp'
|
||||
runs-on: 'ubuntu-latest'
|
||||
base-image: "ubuntu:24.04"
|
||||
skip-drivers: 'false'
|
||||
backend: "moss-transcribe-cpp"
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'cublas'
|
||||
cuda-major-version: "13"
|
||||
cuda-minor-version: "0"
|
||||
@@ -1760,6 +1883,19 @@ include:
|
||||
backend: "parakeet-cpp"
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
- build-type: 'cublas'
|
||||
cuda-major-version: "13"
|
||||
cuda-minor-version: "0"
|
||||
platforms: 'linux/arm64'
|
||||
skip-drivers: 'false'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-nvidia-l4t-cuda-13-arm64-moss-transcribe-cpp'
|
||||
base-image: "ubuntu:24.04"
|
||||
ubuntu-version: '2404'
|
||||
runs-on: 'ubuntu-24.04-arm'
|
||||
backend: "moss-transcribe-cpp"
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
- build-type: 'cublas'
|
||||
cuda-major-version: "13"
|
||||
cuda-minor-version: "0"
|
||||
@@ -1786,6 +1922,19 @@ include:
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'cublas'
|
||||
cuda-major-version: "13"
|
||||
cuda-minor-version: "0"
|
||||
platforms: 'linux/amd64'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-gpu-nvidia-cuda-13-moss-tts-cpp'
|
||||
runs-on: 'ubuntu-latest'
|
||||
base-image: "ubuntu:24.04"
|
||||
skip-drivers: 'false'
|
||||
backend: "moss-tts-cpp"
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'cublas'
|
||||
cuda-major-version: "13"
|
||||
cuda-minor-version: "0"
|
||||
@@ -1838,6 +1987,19 @@ include:
|
||||
backend: "qwen3-tts-cpp"
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
- build-type: 'cublas'
|
||||
cuda-major-version: "13"
|
||||
cuda-minor-version: "0"
|
||||
platforms: 'linux/arm64'
|
||||
skip-drivers: 'false'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-nvidia-l4t-cuda-13-arm64-moss-tts-cpp'
|
||||
base-image: "ubuntu:24.04"
|
||||
ubuntu-version: '2404'
|
||||
runs-on: 'ubuntu-24.04-arm'
|
||||
backend: "moss-tts-cpp"
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
- build-type: 'cublas'
|
||||
cuda-major-version: "13"
|
||||
cuda-minor-version: "0"
|
||||
@@ -1905,6 +2067,20 @@ include:
|
||||
dockerfile: "./backend/Dockerfile.llama-cpp"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'hipblas'
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
platforms: 'linux/amd64'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-gpu-rocm-hipblas-bonsai'
|
||||
builder-base-image: 'quay.io/go-skynet/ci-cache:base-grpc-rocm-amd64'
|
||||
runs-on: 'ubuntu-latest'
|
||||
base-image: "rocm/dev-ubuntu-24.04:7.2.1"
|
||||
skip-drivers: 'false'
|
||||
backend: "bonsai"
|
||||
dockerfile: "./backend/Dockerfile.bonsai"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'hipblas'
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
@@ -2169,6 +2345,20 @@ include:
|
||||
dockerfile: "./backend/Dockerfile.turboquant"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'sycl_f32'
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
platforms: 'linux/amd64'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-gpu-intel-sycl-f32-bonsai'
|
||||
builder-base-image: 'quay.io/go-skynet/ci-cache:base-grpc-intel-amd64'
|
||||
runs-on: 'ubuntu-latest'
|
||||
base-image: "intel/oneapi-basekit:2025.3.0-0-devel-ubuntu24.04"
|
||||
skip-drivers: 'false'
|
||||
backend: "bonsai"
|
||||
dockerfile: "./backend/Dockerfile.bonsai"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'sycl_f16'
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
@@ -2197,6 +2387,20 @@ include:
|
||||
dockerfile: "./backend/Dockerfile.turboquant"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'sycl_f16'
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
platforms: 'linux/amd64'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-gpu-intel-sycl-f16-bonsai'
|
||||
builder-base-image: 'quay.io/go-skynet/ci-cache:base-grpc-intel-amd64'
|
||||
runs-on: 'ubuntu-latest'
|
||||
base-image: "intel/oneapi-basekit:2025.3.0-0-devel-ubuntu24.04"
|
||||
skip-drivers: 'false'
|
||||
backend: "bonsai"
|
||||
dockerfile: "./backend/Dockerfile.bonsai"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'intel'
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
@@ -2649,6 +2853,21 @@ include:
|
||||
dockerfile: "./backend/Dockerfile.turboquant"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: ''
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
platforms: 'linux/amd64'
|
||||
platform-tag: 'amd64'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-cpu-bonsai'
|
||||
builder-base-image: 'quay.io/go-skynet/ci-cache:base-grpc-amd64'
|
||||
runs-on: 'ubuntu-latest'
|
||||
base-image: "ubuntu:24.04"
|
||||
skip-drivers: 'false'
|
||||
backend: "bonsai"
|
||||
dockerfile: "./backend/Dockerfile.bonsai"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: ''
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
@@ -2664,6 +2883,21 @@ include:
|
||||
dockerfile: "./backend/Dockerfile.turboquant"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: ''
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
platforms: 'linux/arm64'
|
||||
platform-tag: 'arm64'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-cpu-bonsai'
|
||||
builder-base-image: 'quay.io/go-skynet/ci-cache:base-grpc-arm64'
|
||||
runs-on: 'ubuntu-24.04-arm'
|
||||
base-image: "ubuntu:24.04"
|
||||
skip-drivers: 'false'
|
||||
backend: "bonsai"
|
||||
dockerfile: "./backend/Dockerfile.bonsai"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: ''
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
@@ -2806,6 +3040,20 @@ include:
|
||||
dockerfile: "./backend/Dockerfile.turboquant"
|
||||
context: "./"
|
||||
ubuntu-version: '2204'
|
||||
- build-type: 'cublas'
|
||||
cuda-major-version: "12"
|
||||
cuda-minor-version: "0"
|
||||
platforms: 'linux/arm64'
|
||||
skip-drivers: 'false'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-nvidia-l4t-arm64-bonsai'
|
||||
builder-base-image: 'quay.io/go-skynet/ci-cache:base-grpc-l4t-cuda-12-arm64'
|
||||
base-image: "nvcr.io/nvidia/l4t-jetpack:r36.4.0"
|
||||
runs-on: 'ubuntu-24.04-arm'
|
||||
backend: "bonsai"
|
||||
dockerfile: "./backend/Dockerfile.bonsai"
|
||||
context: "./"
|
||||
ubuntu-version: '2204'
|
||||
- build-type: 'vulkan'
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
@@ -2852,6 +3100,22 @@ include:
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
# Stablediffusion-ggml
|
||||
- build-type: 'vulkan'
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
platforms: 'linux/amd64'
|
||||
platform-tag: 'amd64'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-gpu-vulkan-bonsai'
|
||||
builder-base-image: 'quay.io/go-skynet/ci-cache:base-grpc-vulkan-amd64'
|
||||
runs-on: 'ubuntu-latest'
|
||||
base-image: "ubuntu:24.04"
|
||||
skip-drivers: 'false'
|
||||
backend: "bonsai"
|
||||
dockerfile: "./backend/Dockerfile.bonsai"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
# Stablediffusion-ggml
|
||||
- build-type: 'vulkan'
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
@@ -2868,6 +3132,22 @@ include:
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
# Stablediffusion-ggml
|
||||
- build-type: 'vulkan'
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
platforms: 'linux/arm64'
|
||||
platform-tag: 'arm64'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-gpu-vulkan-bonsai'
|
||||
builder-base-image: 'quay.io/go-skynet/ci-cache:base-grpc-vulkan-arm64'
|
||||
runs-on: 'ubuntu-24.04-arm'
|
||||
base-image: "ubuntu:24.04"
|
||||
skip-drivers: 'false'
|
||||
backend: "bonsai"
|
||||
dockerfile: "./backend/Dockerfile.bonsai"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
# Stablediffusion-ggml
|
||||
- build-type: ''
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
@@ -3597,6 +3877,115 @@ include:
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
# moss-transcribe-cpp
|
||||
- build-type: ''
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
platforms: 'linux/amd64'
|
||||
platform-tag: 'amd64'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-cpu-moss-transcribe-cpp'
|
||||
runs-on: 'ubuntu-latest'
|
||||
base-image: "ubuntu:24.04"
|
||||
skip-drivers: 'false'
|
||||
backend: "moss-transcribe-cpp"
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: ''
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
platforms: 'linux/arm64'
|
||||
platform-tag: 'arm64'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-cpu-moss-transcribe-cpp'
|
||||
runs-on: 'ubuntu-24.04-arm'
|
||||
base-image: "ubuntu:24.04"
|
||||
skip-drivers: 'false'
|
||||
backend: "moss-transcribe-cpp"
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'sycl_f32'
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
platforms: 'linux/amd64'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-gpu-intel-sycl-f32-moss-transcribe-cpp'
|
||||
runs-on: 'ubuntu-latest'
|
||||
base-image: "intel/oneapi-basekit:2025.3.0-0-devel-ubuntu24.04"
|
||||
skip-drivers: 'false'
|
||||
backend: "moss-transcribe-cpp"
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'sycl_f16'
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
platforms: 'linux/amd64'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-gpu-intel-sycl-f16-moss-transcribe-cpp'
|
||||
runs-on: 'ubuntu-latest'
|
||||
base-image: "intel/oneapi-basekit:2025.3.0-0-devel-ubuntu24.04"
|
||||
skip-drivers: 'false'
|
||||
backend: "moss-transcribe-cpp"
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'vulkan'
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
platforms: 'linux/amd64'
|
||||
platform-tag: 'amd64'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-gpu-vulkan-moss-transcribe-cpp'
|
||||
runs-on: 'ubuntu-latest'
|
||||
base-image: "ubuntu:24.04"
|
||||
skip-drivers: 'false'
|
||||
backend: "moss-transcribe-cpp"
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'vulkan'
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
platforms: 'linux/arm64'
|
||||
platform-tag: 'arm64'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-gpu-vulkan-moss-transcribe-cpp'
|
||||
runs-on: 'ubuntu-24.04-arm'
|
||||
base-image: "ubuntu:24.04"
|
||||
skip-drivers: 'false'
|
||||
backend: "moss-transcribe-cpp"
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'cublas'
|
||||
cuda-major-version: "12"
|
||||
cuda-minor-version: "0"
|
||||
platforms: 'linux/arm64'
|
||||
skip-drivers: 'false'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-nvidia-l4t-arm64-moss-transcribe-cpp'
|
||||
base-image: "nvcr.io/nvidia/l4t-jetpack:r36.4.0"
|
||||
runs-on: 'ubuntu-24.04-arm'
|
||||
backend: "moss-transcribe-cpp"
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2204'
|
||||
- build-type: 'hipblas'
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
platforms: 'linux/amd64'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-gpu-rocm-hipblas-moss-transcribe-cpp'
|
||||
base-image: "rocm/dev-ubuntu-24.04:7.2.1"
|
||||
runs-on: 'ubuntu-latest'
|
||||
skip-drivers: 'false'
|
||||
backend: "moss-transcribe-cpp"
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
# ced
|
||||
- build-type: 'cublas'
|
||||
cuda-major-version: "12"
|
||||
@@ -4179,6 +4568,35 @@ include:
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
# moss-tts-cpp
|
||||
- build-type: ''
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
platforms: 'linux/amd64'
|
||||
platform-tag: 'amd64'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-cpu-moss-tts-cpp'
|
||||
runs-on: 'ubuntu-latest'
|
||||
base-image: "ubuntu:24.04"
|
||||
skip-drivers: 'false'
|
||||
backend: "moss-tts-cpp"
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: ''
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
platforms: 'linux/arm64'
|
||||
platform-tag: 'arm64'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-cpu-moss-tts-cpp'
|
||||
runs-on: 'ubuntu-24.04-arm'
|
||||
base-image: "ubuntu:24.04"
|
||||
skip-drivers: 'false'
|
||||
backend: "moss-tts-cpp"
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
# omnivoice-cpp
|
||||
- build-type: ''
|
||||
cuda-major-version: ""
|
||||
@@ -4221,6 +4639,19 @@ include:
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'sycl_f32'
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
platforms: 'linux/amd64'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-gpu-intel-sycl-f32-moss-tts-cpp'
|
||||
runs-on: 'ubuntu-latest'
|
||||
base-image: "intel/oneapi-basekit:2025.3.0-0-devel-ubuntu24.04"
|
||||
skip-drivers: 'false'
|
||||
backend: "moss-tts-cpp"
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'sycl_f32'
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
@@ -4247,6 +4678,19 @@ include:
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'sycl_f16'
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
platforms: 'linux/amd64'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-gpu-intel-sycl-f16-moss-tts-cpp'
|
||||
runs-on: 'ubuntu-latest'
|
||||
base-image: "intel/oneapi-basekit:2025.3.0-0-devel-ubuntu24.04"
|
||||
skip-drivers: 'false'
|
||||
backend: "moss-tts-cpp"
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'sycl_f16'
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
@@ -4274,6 +4718,20 @@ include:
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'vulkan'
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
platforms: 'linux/amd64'
|
||||
platform-tag: 'amd64'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-gpu-vulkan-moss-tts-cpp'
|
||||
runs-on: 'ubuntu-latest'
|
||||
base-image: "ubuntu:24.04"
|
||||
skip-drivers: 'false'
|
||||
backend: "moss-tts-cpp"
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'vulkan'
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
@@ -4302,6 +4760,20 @@ include:
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'vulkan'
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
platforms: 'linux/arm64'
|
||||
platform-tag: 'arm64'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-gpu-vulkan-moss-tts-cpp'
|
||||
runs-on: 'ubuntu-24.04-arm'
|
||||
base-image: "ubuntu:24.04"
|
||||
skip-drivers: 'false'
|
||||
backend: "moss-tts-cpp"
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'vulkan'
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
@@ -4329,6 +4801,19 @@ include:
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2204'
|
||||
- build-type: 'cublas'
|
||||
cuda-major-version: "12"
|
||||
cuda-minor-version: "0"
|
||||
platforms: 'linux/arm64'
|
||||
skip-drivers: 'false'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-nvidia-l4t-arm64-moss-tts-cpp'
|
||||
base-image: "nvcr.io/nvidia/l4t-jetpack:r36.4.0"
|
||||
runs-on: 'ubuntu-24.04-arm'
|
||||
backend: "moss-tts-cpp"
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2204'
|
||||
- build-type: 'cublas'
|
||||
cuda-major-version: "12"
|
||||
cuda-minor-version: "0"
|
||||
@@ -4355,6 +4840,19 @@ include:
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'hipblas'
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
platforms: 'linux/amd64'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-gpu-rocm-hipblas-moss-tts-cpp'
|
||||
base-image: "rocm/dev-ubuntu-24.04:6.4.4"
|
||||
runs-on: 'ubuntu-latest'
|
||||
skip-drivers: 'false'
|
||||
backend: "moss-tts-cpp"
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'hipblas'
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
@@ -4652,7 +5150,6 @@ include:
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
# rfdetr
|
||||
- build-type: ''
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
@@ -4667,6 +5164,35 @@ include:
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
# cloud-proxy
|
||||
- build-type: ''
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
platforms: 'linux/amd64'
|
||||
platform-tag: 'amd64'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-cpu-cloud-proxy'
|
||||
runs-on: 'ubuntu-latest'
|
||||
base-image: "ubuntu:24.04"
|
||||
skip-drivers: 'false'
|
||||
backend: "cloud-proxy"
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: ''
|
||||
cuda-major-version: ""
|
||||
cuda-minor-version: ""
|
||||
platforms: 'linux/arm64'
|
||||
platform-tag: 'arm64'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-cpu-cloud-proxy'
|
||||
runs-on: 'ubuntu-24.04-arm'
|
||||
base-image: "ubuntu:24.04"
|
||||
skip-drivers: 'false'
|
||||
backend: "cloud-proxy"
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
# rfdetr
|
||||
- build-type: ''
|
||||
cuda-major-version: ""
|
||||
@@ -5177,39 +5703,6 @@ include:
|
||||
dockerfile: "./backend/Dockerfile.golang"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
# llama-cpp-localai-paged: the LocalAI paged-attention llama.cpp variant. Each
|
||||
# row mirrors the corresponding llama-cpp row with backend/dockerfile/tag-suffix
|
||||
# swapped; builder-base-image is left UNCHANGED so these reuse the same
|
||||
# base-grpc-* prebuilt bases (same gRPC + same toolchain), needing no new
|
||||
# base-images.yml variant.
|
||||
- build-type: 'cublas'
|
||||
cuda-major-version: "13"
|
||||
cuda-minor-version: "0"
|
||||
platforms: 'linux/amd64'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-gpu-nvidia-cuda-13-llama-cpp-localai-paged'
|
||||
builder-base-image: 'quay.io/go-skynet/ci-cache:base-grpc-cuda-13-amd64'
|
||||
runs-on: 'bigger-runner'
|
||||
base-image: "ubuntu:24.04"
|
||||
skip-drivers: 'false'
|
||||
backend: "llama-cpp-localai-paged"
|
||||
dockerfile: "./backend/Dockerfile.llama-cpp-localai-paged"
|
||||
context: "./"
|
||||
ubuntu-version: '2404'
|
||||
- build-type: 'cublas'
|
||||
cuda-major-version: "13"
|
||||
cuda-minor-version: "0"
|
||||
platforms: 'linux/arm64'
|
||||
skip-drivers: 'false'
|
||||
tag-latest: 'auto'
|
||||
tag-suffix: '-nvidia-l4t-cuda-13-arm64-llama-cpp-localai-paged'
|
||||
builder-base-image: 'quay.io/go-skynet/ci-cache:base-grpc-cuda-13-arm64'
|
||||
base-image: "ubuntu:24.04"
|
||||
runs-on: 'ubuntu-24.04-arm'
|
||||
ubuntu-version: '2404'
|
||||
backend: "llama-cpp-localai-paged"
|
||||
dockerfile: "./backend/Dockerfile.llama-cpp-localai-paged"
|
||||
context: "./"
|
||||
|
||||
# Darwin matrix (consumed by backend-jobs-darwin).
|
||||
includeDarwin:
|
||||
@@ -5253,6 +5746,10 @@ includeDarwin:
|
||||
tag-suffix: "-metal-darwin-arm64-parakeet-cpp"
|
||||
build-type: "metal"
|
||||
lang: "go"
|
||||
- backend: "moss-transcribe-cpp"
|
||||
tag-suffix: "-metal-darwin-arm64-moss-transcribe-cpp"
|
||||
build-type: "metal"
|
||||
lang: "go"
|
||||
- backend: "ced"
|
||||
tag-suffix: "-metal-darwin-arm64-ced"
|
||||
build-type: "metal"
|
||||
@@ -5273,6 +5770,10 @@ includeDarwin:
|
||||
tag-suffix: "-metal-darwin-arm64-qwen3-tts-cpp"
|
||||
build-type: "metal"
|
||||
lang: "go"
|
||||
- backend: "moss-tts-cpp"
|
||||
tag-suffix: "-metal-darwin-arm64-moss-tts-cpp"
|
||||
build-type: "metal"
|
||||
lang: "go"
|
||||
- backend: "omnivoice-cpp"
|
||||
tag-suffix: "-metal-darwin-arm64-omnivoice-cpp"
|
||||
build-type: "metal"
|
||||
@@ -5398,6 +5899,10 @@ includeDarwin:
|
||||
tag-suffix: "-metal-darwin-arm64-local-store"
|
||||
build-type: "metal"
|
||||
lang: "go"
|
||||
- backend: "cloud-proxy"
|
||||
tag-suffix: "-metal-darwin-arm64-cloud-proxy"
|
||||
build-type: "metal"
|
||||
lang: "go"
|
||||
- backend: "llama-cpp-quantization"
|
||||
tag-suffix: "-metal-darwin-arm64-llama-cpp-quantization"
|
||||
build-type: "mps"
|
||||
|
||||
14
.github/bump_deps.sh
vendored
14
.github/bump_deps.sh
vendored
@@ -9,7 +9,19 @@ if [ -z "$FILE" ]; then
|
||||
FILE="Makefile"
|
||||
fi
|
||||
|
||||
LAST_COMMIT=$(curl -s -H "Accept: application/vnd.github.VERSION.sha" "https://api.github.com/repos/$REPO/commits/$BRANCH")
|
||||
# -L so a renamed/transferred upstream repo (GitHub answers 301) still
|
||||
# resolves instead of handing us the redirect body, and -f so an HTTP error
|
||||
# aborts the run rather than letting an error page reach sed below.
|
||||
LAST_COMMIT=$(curl -sfL -H "Accept: application/vnd.github.VERSION.sha" "https://api.github.com/repos/$REPO/commits/$BRANCH")
|
||||
|
||||
# Guard the sed input: anything that is not a bare 40-hex SHA (an API error
|
||||
# body, an empty response) would otherwise be spliced into the Makefile pin —
|
||||
# either corrupting it silently or blowing up sed with an unterminated
|
||||
# expression, which is how this job failed for a renamed repo.
|
||||
if ! [[ "$LAST_COMMIT" =~ ^[0-9a-f]{40}$ ]]; then
|
||||
echo "Refusing to bump $VAR: expected a 40-char commit SHA for $REPO@$BRANCH, got: $LAST_COMMIT" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Read $VAR from Makefile (only first match)
|
||||
set +e
|
||||
|
||||
133
.github/ci/variantproposals/body.go
vendored
Normal file
133
.github/ci/variantproposals/body.go
vendored
Normal file
@@ -0,0 +1,133 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// RenderBody writes the pull request body.
|
||||
//
|
||||
// The body is the product of this job, not the diff. Grouping is a judgement
|
||||
// call that has gone wrong in both directions before, so a reviewer has to be
|
||||
// able to accept or reject each family from the body alone, without opening
|
||||
// HuggingFace to work out whether two entries hold the same weights.
|
||||
func RenderBody(r *Result, ledgerPath string) string {
|
||||
var b strings.Builder
|
||||
|
||||
b.WriteString("## Proposed gallery variant groupings\n\n")
|
||||
b.WriteString("This is a proposal, not a decision. The gallery agent adds one build per model and never joins an existing family, so entries that are alternative builds of the same weights drift apart as the gallery grows. This job re-applies the grouping heuristics from the manual sweeps and asks a human to confirm.\n\n")
|
||||
b.WriteString("Each family below lists the parent, the variants, and the evidence that they are the same weights. **Reject anything whose evidence you do not believe.**\n\n")
|
||||
b.WriteString(fmt.Sprintf("To decline a family permanently, add one line to `%s` in this pull request and close it:\n\n", ledgerPath))
|
||||
b.WriteString("```yaml\npairs:\n - {parent: some-model, variant: some-model-thing, reason: \"different finetune\"}\n```\n\n")
|
||||
|
||||
b.WriteString(fmt.Sprintf("### Proposed families (%d)\n\n", len(r.Families)))
|
||||
if len(r.Families) == 0 {
|
||||
b.WriteString("None.\n\n")
|
||||
}
|
||||
for _, f := range r.Families {
|
||||
b.WriteString(fmt.Sprintf("#### `%s`\n\n", f.Parent))
|
||||
b.WriteString("| variant | signals | evidence |\n|---|---|---|\n")
|
||||
for _, p := range f.Proposals {
|
||||
b.WriteString(fmt.Sprintf("| `%s` | %s | %s |\n", p.Variant, joinSignals(p.Evidence.Signals), describeEvidence(p.Evidence)))
|
||||
}
|
||||
b.WriteString("\n")
|
||||
}
|
||||
|
||||
b.WriteString(fmt.Sprintf("### Declined by the ledger (%d)\n\n", len(r.Suppressed)))
|
||||
if len(r.Suppressed) == 0 {
|
||||
b.WriteString("Nothing the heuristics found was already on the ledger.\n\n")
|
||||
} else {
|
||||
b.WriteString("Candidates the heuristics found and the ledger has already settled. They are listed so the ledger's effect stays visible rather than silently shrinking the job's output.\n\n")
|
||||
for _, s := range r.Suppressed {
|
||||
b.WriteString(fmt.Sprintf("- `%s` + `%s`: %s\n", s.A, s.B, s.Reason))
|
||||
}
|
||||
b.WriteString("\n")
|
||||
}
|
||||
|
||||
if len(r.AliasSkipped) > 0 {
|
||||
b.WriteString(fmt.Sprintf("### Aliases, not variants (%d)\n\n", len(r.AliasSkipped)))
|
||||
b.WriteString("These entries install byte for byte the same payload. An alias exists so clients can send a particular name; folding it under another entry would hide that name.\n\n")
|
||||
for _, s := range r.AliasSkipped {
|
||||
b.WriteString(fmt.Sprintf("- `%s` + `%s`: %s\n", s.A, s.B, s.Reason))
|
||||
}
|
||||
b.WriteString("\n")
|
||||
}
|
||||
|
||||
if len(r.Refusals) > 0 {
|
||||
b.WriteString(fmt.Sprintf("### Found but refused (%d)\n\n", len(r.Refusals)))
|
||||
b.WriteString("Candidates the heuristics found but the authoring rules would not let this job write. They need a human edit or a rule change.\n\n")
|
||||
for _, ref := range r.Refusals {
|
||||
b.WriteString(fmt.Sprintf("- %s: %s\n", codeList(ref.Members), ref.Reason))
|
||||
}
|
||||
b.WriteString("\n")
|
||||
}
|
||||
|
||||
b.WriteString("---\n\nOpened by `.github/ci/variantproposals`. Heuristics and the rejection ledger live there and in the ledger file; a wrong proposal is a bug in one of the two.\n")
|
||||
return b.String()
|
||||
}
|
||||
|
||||
func joinSignals(signals []Signal) string {
|
||||
if len(signals) == 0 {
|
||||
return "inferred through another member of the family"
|
||||
}
|
||||
out := make([]string, 0, len(signals))
|
||||
for _, s := range signals {
|
||||
out = append(out, "`"+string(s)+"`")
|
||||
}
|
||||
return strings.Join(out, ", ")
|
||||
}
|
||||
|
||||
func describeEvidence(e Evidence) string {
|
||||
var parts []string
|
||||
if e.SharedStem != "" {
|
||||
parts = append(parts, fmt.Sprintf("same name once quantization markers are stripped: `%s`", e.SharedStem))
|
||||
}
|
||||
if e.SharedFile != "" {
|
||||
parts = append(parts, fmt.Sprintf("same primary weight filename once quantization markers are stripped: `%s`", e.SharedFile))
|
||||
}
|
||||
if e.SharedRepo != "" {
|
||||
parts = append(parts, fmt.Sprintf("same upstream repo `%s`", e.SharedRepo))
|
||||
}
|
||||
if len(e.QuantTokens) > 0 {
|
||||
parts = append(parts, "differing quantization tokens: `"+strings.Join(e.QuantTokens, "`, `")+"`")
|
||||
}
|
||||
if len(parts) == 0 {
|
||||
return "reached this family through another member"
|
||||
}
|
||||
return strings.Join(parts, "; ")
|
||||
}
|
||||
|
||||
func codeList(names []string) string {
|
||||
out := make([]string, 0, len(names))
|
||||
for _, n := range names {
|
||||
out = append(out, "`"+n+"`")
|
||||
}
|
||||
return strings.Join(out, " + ")
|
||||
}
|
||||
|
||||
// RenderSummary is the terminal-facing digest of a run, so the workflow log
|
||||
// says what happened without anyone opening the pull request.
|
||||
func RenderSummary(r *Result) string {
|
||||
var b strings.Builder
|
||||
fmt.Fprintf(&b, "families proposed: %d\n", len(r.Families))
|
||||
for _, f := range r.Families {
|
||||
names := make([]string, 0, len(f.Proposals))
|
||||
for _, p := range f.Proposals {
|
||||
names = append(names, p.Variant)
|
||||
}
|
||||
fmt.Fprintf(&b, " %s <- %s\n", f.Parent, strings.Join(names, ", "))
|
||||
}
|
||||
fmt.Fprintf(&b, "declined by ledger: %d\n", len(r.Suppressed))
|
||||
for _, s := range r.Suppressed {
|
||||
fmt.Fprintf(&b, " %s\n", s)
|
||||
}
|
||||
fmt.Fprintf(&b, "aliases skipped: %d\n", len(r.AliasSkipped))
|
||||
for _, s := range r.AliasSkipped {
|
||||
fmt.Fprintf(&b, " %s\n", s)
|
||||
}
|
||||
fmt.Fprintf(&b, "refused: %d\n", len(r.Refusals))
|
||||
for _, ref := range r.Refusals {
|
||||
fmt.Fprintf(&b, " %s: %s\n", strings.Join(ref.Members, " + "), ref.Reason)
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
120
.github/ci/variantproposals/edit.go
vendored
Normal file
120
.github/ci/variantproposals/edit.go
vendored
Normal file
@@ -0,0 +1,120 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"regexp"
|
||||
"sort"
|
||||
"strings"
|
||||
)
|
||||
|
||||
var (
|
||||
inlineName = regexp.MustCompile(`^- (?:&\S+ )?name:`)
|
||||
keyName = regexp.MustCompile(`^ name:`)
|
||||
keyVariants = regexp.MustCompile(`^ variants:\s*(.*)$`)
|
||||
variantItem = regexp.MustCompile(`^ - `)
|
||||
unsafeInName = regexp.MustCompile(`[:#{}\[\],&*?|>'"%@` + "`" + `]|^\s|\s$`)
|
||||
)
|
||||
|
||||
// ApplyFamilies writes the proposed variant lists into the index text.
|
||||
//
|
||||
// The edit is textual on purpose. Re-serialising the index through a YAML
|
||||
// marshaller would reflow 40,000 lines, drop the anchors and merge keys the
|
||||
// gallery relies on, and produce a diff no reviewer could read, which would
|
||||
// make the pull request worthless even when the proposals inside it are right.
|
||||
func ApplyFamilies(ix *Index, families []Family) ([]string, error) {
|
||||
byName, _ := ix.ByName()
|
||||
|
||||
type edit struct {
|
||||
at int
|
||||
remove int
|
||||
insert []string
|
||||
ordinal int
|
||||
}
|
||||
var edits []edit
|
||||
|
||||
for _, f := range families {
|
||||
entry, ok := byName[strings.ToLower(f.Parent)]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("parent %q is not in the index", f.Parent)
|
||||
}
|
||||
items := make([]string, 0, len(f.Proposals))
|
||||
for _, p := range f.Proposals {
|
||||
items = append(items, " - model: "+quoteName(p.Variant))
|
||||
}
|
||||
|
||||
at, remove, err := insertionPoint(ix, entry)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
insert := items
|
||||
if remove > 0 || !hasVariantsKey(ix, entry) {
|
||||
insert = append([]string{" variants:"}, items...)
|
||||
}
|
||||
edits = append(edits, edit{at: at, remove: remove, insert: insert, ordinal: entry.Index})
|
||||
}
|
||||
|
||||
// Applying from the bottom up keeps every line number computed against the
|
||||
// original text valid while earlier edits are still pending.
|
||||
sort.Slice(edits, func(i, j int) bool { return edits[i].at > edits[j].at })
|
||||
|
||||
lines := append([]string(nil), ix.Lines...)
|
||||
for _, e := range edits {
|
||||
tail := append([]string(nil), lines[e.at+e.remove:]...)
|
||||
lines = append(lines[:e.at], append(append([]string(nil), e.insert...), tail...)...)
|
||||
}
|
||||
return lines, nil
|
||||
}
|
||||
|
||||
func hasVariantsKey(ix *Index, e *GalleryEntry) bool {
|
||||
for i := e.StartLine; i < e.EndLine; i++ {
|
||||
if keyVariants.MatchString(ix.Lines[i]) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// insertionPoint reports where new variant items belong, and how many existing
|
||||
// lines the insertion replaces.
|
||||
//
|
||||
// An entry with no variants key gets one right after its name, which is where
|
||||
// the hand-written families put it. An entry with an empty "variants: []" has
|
||||
// that line replaced by a block. An entry with a block gets its items appended.
|
||||
func insertionPoint(ix *Index, e *GalleryEntry) (at int, remove int, err error) {
|
||||
for i := e.StartLine; i < e.EndLine; i++ {
|
||||
m := keyVariants.FindStringSubmatch(ix.Lines[i])
|
||||
if m == nil {
|
||||
continue
|
||||
}
|
||||
if strings.TrimSpace(m[1]) == "[]" {
|
||||
return i, 1, nil
|
||||
}
|
||||
if strings.TrimSpace(m[1]) != "" {
|
||||
return 0, 0, fmt.Errorf("entry %q writes its variants inline (%q); this job only edits block lists", e.Name, strings.TrimSpace(m[1]))
|
||||
}
|
||||
last := i
|
||||
for j := i + 1; j < e.EndLine && variantItem.MatchString(ix.Lines[j]); j++ {
|
||||
last = j
|
||||
}
|
||||
return last + 1, 0, nil
|
||||
}
|
||||
|
||||
if inlineName.MatchString(ix.Lines[e.StartLine]) {
|
||||
return e.StartLine + 1, 0, nil
|
||||
}
|
||||
for i := e.StartLine; i < e.EndLine; i++ {
|
||||
if keyName.MatchString(ix.Lines[i]) {
|
||||
return i + 1, 0, nil
|
||||
}
|
||||
}
|
||||
return 0, 0, fmt.Errorf("entry %q has no name line to anchor the insertion to", e.Name)
|
||||
}
|
||||
|
||||
// quoteName quotes a variant reference when the name would otherwise change
|
||||
// meaning as bare YAML. Config-suffixed names carry a ":" and always need it.
|
||||
func quoteName(name string) string {
|
||||
if unsafeInName.MatchString(name) {
|
||||
return `"` + strings.ReplaceAll(name, `"`, `\"`) + `"`
|
||||
}
|
||||
return name
|
||||
}
|
||||
153
.github/ci/variantproposals/edit_test.go
vendored
Normal file
153
.github/ci/variantproposals/edit_test.go
vendored
Normal file
@@ -0,0 +1,153 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"strings"
|
||||
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
)
|
||||
|
||||
var _ = Describe("ApplyFamilies", func() {
|
||||
apply := func(ix *Index, families []Family) []string {
|
||||
lines, err := ApplyFamilies(ix, families)
|
||||
ExpectWithOffset(1, err).ToNot(HaveOccurred())
|
||||
return lines
|
||||
}
|
||||
|
||||
// insertedLines is what a reviewer would see in the diff. A textual editor
|
||||
// that reflowed the file would show thousands here, which is the failure
|
||||
// this whole approach exists to avoid.
|
||||
insertedLines := func(before, after []string) int {
|
||||
remaining := map[string]int{}
|
||||
for _, l := range before {
|
||||
remaining[l]++
|
||||
}
|
||||
n := 0
|
||||
for _, l := range after {
|
||||
if remaining[l] > 0 {
|
||||
remaining[l]--
|
||||
continue
|
||||
}
|
||||
n++
|
||||
}
|
||||
return n
|
||||
}
|
||||
|
||||
It("adds a variants block right after the entry's name and touches nothing else", func() {
|
||||
ix := indexOf(
|
||||
entryYAML("foo-model", "acme/repo", "foo-model-Q4_K_M.gguf", "aa"),
|
||||
entryYAML("foo-model-q8_0", "acme/repo", "foo-model-Q8_0.gguf", "bb"),
|
||||
)
|
||||
out := apply(ix, []Family{{Parent: "foo-model", Proposals: []Proposal{{Variant: "foo-model-q8_0"}}}})
|
||||
|
||||
Expect(out[0]).To(Equal("- name: foo-model"))
|
||||
Expect(out[1]).To(Equal(" variants:"))
|
||||
Expect(out[2]).To(Equal(" - model: foo-model-q8_0"))
|
||||
Expect(len(out)).To(Equal(len(ix.Lines) + 2))
|
||||
Expect(insertedLines(ix.Lines, out)).To(Equal(2))
|
||||
})
|
||||
|
||||
It("appends to a variants block that already exists", func() {
|
||||
ix := indexOf(`- name: partial
|
||||
variants:
|
||||
- model: partial-q8_0
|
||||
url: u
|
||||
overrides:
|
||||
parameters:
|
||||
model: partial-Q4_K_M.gguf
|
||||
`, entryYAML("partial-f16", "acme/repo", "partial-f16.gguf", "cc"))
|
||||
out := apply(ix, []Family{{Parent: "partial", Proposals: []Proposal{{Variant: "partial-f16"}}}})
|
||||
|
||||
Expect(out[1]).To(Equal(" variants:"))
|
||||
Expect(out[2]).To(Equal(" - model: partial-q8_0"))
|
||||
Expect(out[3]).To(Equal(" - model: partial-f16"))
|
||||
Expect(out[4]).To(Equal(" url: u"))
|
||||
})
|
||||
|
||||
It("replaces an explicit empty list rather than leaving two variants keys", func() {
|
||||
ix := indexOf(`- name: emptied
|
||||
variants: []
|
||||
url: u
|
||||
`, entryYAML("emptied-q8_0", "acme/repo", "emptied-Q8_0.gguf", "cc"))
|
||||
out := apply(ix, []Family{{Parent: "emptied", Proposals: []Proposal{{Variant: "emptied-q8_0"}}}})
|
||||
|
||||
Expect(strings.Join(out[:4], "\n")).To(Equal("- name: emptied\n variants:\n - model: emptied-q8_0\n url: u"))
|
||||
Expect(strings.Count(strings.Join(out, "\n"), "variants:")).To(Equal(1))
|
||||
})
|
||||
|
||||
It("quotes a config-suffixed name so the reference stays a string", func() {
|
||||
ix := indexOf(
|
||||
entryYAML("phi-2-chat", "acme/repo", "phi-2-chat-Q4_K_M.gguf", "aa"),
|
||||
entryYAML("phi-2-chat:Q8_0", "acme/repo", "phi-2-chat-Q8_0.gguf", "bb"),
|
||||
)
|
||||
out := apply(ix, []Family{{Parent: "phi-2-chat", Proposals: []Proposal{{Variant: "phi-2-chat:Q8_0"}}}})
|
||||
Expect(out[2]).To(Equal(` - model: "phi-2-chat:Q8_0"`))
|
||||
|
||||
// The result has to still be a gallery, and the reference has to
|
||||
// resolve to the entry it names.
|
||||
reparsed, err := ParseIndex(strings.Join(out, "\n"))
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(reparsed.Entries[0].Variants).To(ConsistOf(VariantRef{Model: "phi-2-chat:Q8_0"}))
|
||||
})
|
||||
|
||||
It("keeps line numbers correct when several entries are edited at once", func() {
|
||||
ix := indexOf(
|
||||
entryYAML("alpha", "acme/repo", "alpha-Q4_K_M.gguf", "aa"),
|
||||
entryYAML("alpha-q8_0", "acme/repo", "alpha-Q8_0.gguf", "bb"),
|
||||
entryYAML("beta", "acme/repo", "beta-Q4_K_M.gguf", "cc"),
|
||||
entryYAML("beta-q8_0", "acme/repo", "beta-Q8_0.gguf", "dd"),
|
||||
)
|
||||
out := apply(ix, []Family{
|
||||
{Parent: "alpha", Proposals: []Proposal{{Variant: "alpha-q8_0"}}},
|
||||
{Parent: "beta", Proposals: []Proposal{{Variant: "beta-q8_0"}}},
|
||||
})
|
||||
|
||||
reparsed, err := ParseIndex(strings.Join(out, "\n"))
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(reparsed.Entries).To(HaveLen(4))
|
||||
Expect(reparsed.Entries[0].Variants).To(ConsistOf(VariantRef{Model: "alpha-q8_0"}))
|
||||
Expect(reparsed.Entries[2].Variants).To(ConsistOf(VariantRef{Model: "beta-q8_0"}))
|
||||
Expect(reparsed.Entries[1].Variants).To(BeEmpty())
|
||||
Expect(reparsed.Entries[3].Variants).To(BeEmpty())
|
||||
})
|
||||
|
||||
It("fails loudly rather than editing an entry it cannot find", func() {
|
||||
ix := indexOf(entryYAML("only", "acme/repo", "only-Q4_K_M.gguf", "aa"))
|
||||
_, err := ApplyFamilies(ix, []Family{{Parent: "missing", Proposals: []Proposal{{Variant: "x"}}}})
|
||||
Expect(err).To(MatchError(ContainSubstring("not in the index")))
|
||||
})
|
||||
})
|
||||
|
||||
var _ = Describe("ParseIndex", func() {
|
||||
It("records the anchor an entry defines and the anchor an entry merges", func() {
|
||||
ix := indexOf(`- &anc
|
||||
name: anchored
|
||||
url: u
|
||||
`, `- !!merge <<: *anc
|
||||
name: child
|
||||
`)
|
||||
Expect(ix.Entries[0].AnchorName).To(Equal("anc"))
|
||||
Expect(ix.Entries[1].MergesFrom).To(Equal("anc"))
|
||||
Expect(ix.MergeChildren("anc")).To(HaveLen(1))
|
||||
})
|
||||
|
||||
It("carries merged values into the child, so an inherited variants key is visible", func() {
|
||||
ix := indexOf(`- &anc
|
||||
name: anchored
|
||||
url: u
|
||||
variants:
|
||||
- model: something
|
||||
`, `- !!merge <<: *anc
|
||||
name: child
|
||||
`)
|
||||
Expect(ix.Entries[1].HasVariants()).To(BeTrue())
|
||||
})
|
||||
|
||||
It("refuses a list item that decodes to nothing", func() {
|
||||
// Every line number the editor works from comes from pairing decoded
|
||||
// entries with top level list items. If those two views can disagree,
|
||||
// the editor writes into the wrong entry, so the parse refuses instead.
|
||||
_, err := ParseIndex("- name: one\n url: u\n-\n")
|
||||
Expect(err).To(MatchError(ContainSubstring("empty")))
|
||||
})
|
||||
})
|
||||
286
.github/ci/variantproposals/index.go
vendored
Normal file
286
.github/ci/variantproposals/index.go
vendored
Normal file
@@ -0,0 +1,286 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"regexp"
|
||||
"sort"
|
||||
"strings"
|
||||
|
||||
"gopkg.in/yaml.v3"
|
||||
)
|
||||
|
||||
// File is the subset of a gallery file entry the proposer reads.
|
||||
type File struct {
|
||||
Filename string `yaml:"filename"`
|
||||
URI string `yaml:"uri"`
|
||||
SHA256 string `yaml:"sha256"`
|
||||
}
|
||||
|
||||
// VariantRef mirrors the gallery's variant reference.
|
||||
type VariantRef struct {
|
||||
Model string `yaml:"model"`
|
||||
}
|
||||
|
||||
// GalleryEntry is one gallery entry, carrying both the semantics the heuristics need
|
||||
// and the text range the editor needs.
|
||||
//
|
||||
// The two views are kept together deliberately. The editor must not round-trip
|
||||
// the index through a YAML marshaller: the gallery is 40,000 lines and a
|
||||
// reflowed diff cannot be reviewed, which defeats the entire point of a job
|
||||
// whose output is a human decision.
|
||||
type GalleryEntry struct {
|
||||
Name string `yaml:"name"`
|
||||
URL string `yaml:"url"`
|
||||
ConfigFile map[string]any `yaml:"config_file"`
|
||||
Overrides map[string]any `yaml:"overrides"`
|
||||
Files []File `yaml:"files"`
|
||||
Variants []VariantRef `yaml:"variants"`
|
||||
|
||||
// Index is the entry's position in gallery order.
|
||||
Index int `yaml:"-"`
|
||||
// StartLine and EndLine bound the entry's lines, zero based and half open.
|
||||
StartLine int `yaml:"-"`
|
||||
EndLine int `yaml:"-"`
|
||||
// AnchorName is set when the entry defines a YAML anchor. Adding a variants
|
||||
// key to such an entry is inherited by everything that merges it, which is
|
||||
// why proposals involving anchors get special treatment.
|
||||
AnchorName string `yaml:"-"`
|
||||
// MergesFrom is the anchor this entry pulls in with "!!merge <<:".
|
||||
MergesFrom string `yaml:"-"`
|
||||
}
|
||||
|
||||
// Index is a parsed gallery index: entries plus the exact lines they came from.
|
||||
type Index struct {
|
||||
Lines []string
|
||||
Entries []*GalleryEntry
|
||||
}
|
||||
|
||||
var (
|
||||
entryStart = regexp.MustCompile(`^-(?: |$)`)
|
||||
anchorStart = regexp.MustCompile(`^- &(\S+)`)
|
||||
mergeStart = regexp.MustCompile(`^- !!merge <<: \*(\S+)`)
|
||||
)
|
||||
|
||||
// LoadIndex reads and parses a gallery index file.
|
||||
func LoadIndex(path string) (*Index, error) {
|
||||
data, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return ParseIndex(string(data))
|
||||
}
|
||||
|
||||
// ParseIndex builds an Index from the raw text of a gallery index.
|
||||
//
|
||||
// The YAML decode and the textual scan are cross checked against each other: if
|
||||
// they disagree on how many entries there are, every line number the editor
|
||||
// would use is suspect, so the run fails rather than editing the wrong entry.
|
||||
func ParseIndex(text string) (*Index, error) {
|
||||
var entries []*GalleryEntry
|
||||
if err := yaml.Unmarshal([]byte(text), &entries); err != nil {
|
||||
return nil, fmt.Errorf("decoding gallery index: %w", err)
|
||||
}
|
||||
|
||||
lines := strings.Split(text, "\n")
|
||||
var starts []int
|
||||
for i, line := range lines {
|
||||
if entryStart.MatchString(line) {
|
||||
starts = append(starts, i)
|
||||
}
|
||||
}
|
||||
if len(starts) != len(entries) {
|
||||
return nil, fmt.Errorf("gallery index has %d decoded entries but %d top level list items; refusing to edit by line number", len(entries), len(starts))
|
||||
}
|
||||
|
||||
for i, e := range entries {
|
||||
if e == nil {
|
||||
return nil, fmt.Errorf("gallery index list item %d is empty; refusing to edit by line number", i)
|
||||
}
|
||||
e.Index = i
|
||||
e.StartLine = starts[i]
|
||||
if i+1 < len(starts) {
|
||||
e.EndLine = starts[i+1]
|
||||
} else {
|
||||
e.EndLine = len(lines)
|
||||
}
|
||||
if m := anchorStart.FindStringSubmatch(lines[e.StartLine]); m != nil {
|
||||
e.AnchorName = m[1]
|
||||
}
|
||||
if m := mergeStart.FindStringSubmatch(lines[e.StartLine]); m != nil {
|
||||
e.MergesFrom = m[1]
|
||||
}
|
||||
}
|
||||
|
||||
return &Index{Lines: lines, Entries: entries}, nil
|
||||
}
|
||||
|
||||
// MergeChildren lists the entries that pull in the given anchor.
|
||||
func (ix *Index) MergeChildren(anchor string) []*GalleryEntry {
|
||||
var out []*GalleryEntry
|
||||
for _, e := range ix.Entries {
|
||||
if e.MergesFrom == anchor {
|
||||
out = append(out, e)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// ByName indexes entries by lowercased name. A name appearing twice keeps the
|
||||
// first occurrence, matching the gallery's own first-match-wins resolution, and
|
||||
// the duplicates are returned so the caller can refuse to touch them: a
|
||||
// proposal naming an ambiguous entry cannot be reviewed.
|
||||
func (ix *Index) ByName() (map[string]*GalleryEntry, map[string]int) {
|
||||
byName := make(map[string]*GalleryEntry, len(ix.Entries))
|
||||
counts := make(map[string]int, len(ix.Entries))
|
||||
for _, e := range ix.Entries {
|
||||
key := strings.ToLower(e.Name)
|
||||
counts[key]++
|
||||
if _, seen := byName[key]; !seen {
|
||||
byName[key] = e
|
||||
}
|
||||
}
|
||||
dupes := map[string]int{}
|
||||
for name, n := range counts {
|
||||
if n > 1 {
|
||||
dupes[name] = n
|
||||
}
|
||||
}
|
||||
return byName, dupes
|
||||
}
|
||||
|
||||
// Installable reports whether installing this entry would put anything on disk.
|
||||
// A variant target that installs nothing is a dead end for the selector, so it
|
||||
// is never proposed as one.
|
||||
func (e *GalleryEntry) Installable() bool {
|
||||
return e.URL != "" || len(e.ConfigFile) > 0 || len(e.Overrides) > 0 || len(e.Files) > 0
|
||||
}
|
||||
|
||||
// HasVariants reports whether the entry already offers builds of its own. Such
|
||||
// an entry cannot be a variant target: nesting is what the gallery's own
|
||||
// resolution refuses.
|
||||
func (e *GalleryEntry) HasVariants() bool {
|
||||
return len(e.Variants) > 0
|
||||
}
|
||||
|
||||
// auxiliaryFile matches the shared side files that several unrelated models
|
||||
// legitimately hand out the same copy of. Grouping on one of these is how an
|
||||
// earlier sweep linked four wan-2.1 entries to each other and Z-Image-Turbo to
|
||||
// qwen3-4b: they shared a text encoder, not weights.
|
||||
var auxiliaryFile = regexp.MustCompile(`(?i)(mmproj|vae|clip|t5|umt5|text_?encoder|tokenizer|\bae\b|^ae\.|scheduler|config)`)
|
||||
|
||||
// IsAuxiliaryFile reports whether a filename is a side file rather than the
|
||||
// model's own weights.
|
||||
func IsAuxiliaryFile(filename string) bool {
|
||||
base := filename
|
||||
if i := strings.LastIndex(base, "/"); i >= 0 {
|
||||
base = base[i+1:]
|
||||
}
|
||||
return auxiliaryFile.MatchString(base)
|
||||
}
|
||||
|
||||
// PrimaryWeightFile returns the filename of the entry's own weights, and
|
||||
// whether one could be identified unambiguously.
|
||||
//
|
||||
// The declared overrides.parameters.model wins because that is the file the
|
||||
// backend is actually pointed at. Falling back to the file list only works when
|
||||
// exactly one non-auxiliary file is present; anything else is ambiguous, and
|
||||
// guessing is precisely the failure mode this heuristic has already had.
|
||||
func (e *GalleryEntry) PrimaryWeightFile() (string, bool) {
|
||||
if params, ok := e.Overrides["parameters"].(map[string]any); ok {
|
||||
if model, ok := params["model"].(string); ok && model != "" && !IsAuxiliaryFile(model) {
|
||||
return model, true
|
||||
}
|
||||
}
|
||||
var candidates []string
|
||||
for _, f := range e.Files {
|
||||
if f.Filename == "" || IsAuxiliaryFile(f.Filename) {
|
||||
continue
|
||||
}
|
||||
candidates = append(candidates, f.Filename)
|
||||
}
|
||||
if len(candidates) == 1 {
|
||||
return candidates[0], true
|
||||
}
|
||||
return "", false
|
||||
}
|
||||
|
||||
// SourceRepo returns the upstream repository the entry's files come from, as a
|
||||
// coarse "host + owner + repo" key.
|
||||
func (e *GalleryEntry) SourceRepo() string {
|
||||
for _, f := range e.Files {
|
||||
if f.URI == "" {
|
||||
continue
|
||||
}
|
||||
return repoKey(f.URI)
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
func repoKey(uri string) string {
|
||||
u := strings.ToLower(uri)
|
||||
u = strings.TrimPrefix(u, "huggingface://")
|
||||
u = strings.TrimPrefix(u, "https://huggingface.co/")
|
||||
u = strings.TrimPrefix(u, "http://huggingface.co/")
|
||||
parts := strings.Split(u, "/")
|
||||
if len(parts) >= 2 {
|
||||
return parts[0] + "/" + parts[1]
|
||||
}
|
||||
return u
|
||||
}
|
||||
|
||||
// SameInstallPayload reports whether two entries install byte for byte the same
|
||||
// thing.
|
||||
//
|
||||
// Entries like this are aliases, not variants. whisper-1 exists so a client
|
||||
// speaking the OpenAI API can send that name and get whisper-base; folding it
|
||||
// under whisper-base as a variant would hide the very name clients send.
|
||||
func SameInstallPayload(a, b *GalleryEntry) bool {
|
||||
if a.URL != b.URL {
|
||||
return false
|
||||
}
|
||||
if !sameYAML(a.Overrides, b.Overrides) || !sameYAML(a.ConfigFile, b.ConfigFile) {
|
||||
return false
|
||||
}
|
||||
return sameChecksums(a.Files, b.Files)
|
||||
}
|
||||
|
||||
func sameChecksums(a, b []File) bool {
|
||||
if len(a) != len(b) || len(a) == 0 {
|
||||
return false
|
||||
}
|
||||
ha := make([]string, 0, len(a))
|
||||
hb := make([]string, 0, len(b))
|
||||
for _, f := range a {
|
||||
if f.SHA256 == "" {
|
||||
return false
|
||||
}
|
||||
ha = append(ha, f.SHA256)
|
||||
}
|
||||
for _, f := range b {
|
||||
if f.SHA256 == "" {
|
||||
return false
|
||||
}
|
||||
hb = append(hb, f.SHA256)
|
||||
}
|
||||
sort.Strings(ha)
|
||||
sort.Strings(hb)
|
||||
for i := range ha {
|
||||
if ha[i] != hb[i] {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
func sameYAML(a, b any) bool {
|
||||
ba, err := yaml.Marshal(a)
|
||||
if err != nil {
|
||||
return false
|
||||
}
|
||||
bb, err := yaml.Marshal(b)
|
||||
if err != nil {
|
||||
return false
|
||||
}
|
||||
return string(ba) == string(bb)
|
||||
}
|
||||
180
.github/ci/variantproposals/ledger.go
vendored
Normal file
180
.github/ci/variantproposals/ledger.go
vendored
Normal file
@@ -0,0 +1,180 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"sort"
|
||||
"strings"
|
||||
|
||||
"gopkg.in/yaml.v3"
|
||||
)
|
||||
|
||||
// Ledger records the grouping decisions a human has already made against the
|
||||
// proposer, so a declined candidate stays declined instead of coming back every
|
||||
// night until reviewers stop reading the job's pull requests.
|
||||
//
|
||||
// It is checked in next to the gallery and is meant to be edited inside the
|
||||
// proposal pull request itself: declining a family is adding one flow-mapping
|
||||
// line under pairs or groups and closing the PR.
|
||||
type Ledger struct {
|
||||
// Tokens are name segments that mark a distinct model rather than another
|
||||
// build of the same one: finetune names, language codes, product suffixes.
|
||||
// A candidate whose two names differ by any of these is never proposed.
|
||||
Tokens []LedgerToken `yaml:"tokens"`
|
||||
// Pairs are individual candidates a human considered and declined. Order
|
||||
// does not matter: the pair is matched both ways round.
|
||||
Pairs []LedgerPair `yaml:"pairs"`
|
||||
// Groups decline every pair drawn from a set at once, for families like a
|
||||
// per-language release where listing each pair would be unreadable.
|
||||
Groups []LedgerGroup `yaml:"groups"`
|
||||
}
|
||||
|
||||
type LedgerToken struct {
|
||||
Token string `yaml:"token"`
|
||||
Reason string `yaml:"reason"`
|
||||
}
|
||||
|
||||
type LedgerPair struct {
|
||||
Parent string `yaml:"parent"`
|
||||
Variant string `yaml:"variant"`
|
||||
Reason string `yaml:"reason"`
|
||||
}
|
||||
|
||||
type LedgerGroup struct {
|
||||
Members []string `yaml:"members"`
|
||||
Reason string `yaml:"reason"`
|
||||
}
|
||||
|
||||
// LoadLedger reads a ledger file. A missing file is not an error: a gallery
|
||||
// that has declined nothing yet is a legitimate state, and failing the job over
|
||||
// it would only teach people to keep an empty file around.
|
||||
func LoadLedger(path string) (*Ledger, error) {
|
||||
data, err := os.ReadFile(path)
|
||||
if os.IsNotExist(err) {
|
||||
return &Ledger{}, nil
|
||||
}
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return ParseLedger(data)
|
||||
}
|
||||
|
||||
func ParseLedger(data []byte) (*Ledger, error) {
|
||||
l := &Ledger{}
|
||||
if err := yaml.Unmarshal(data, l); err != nil {
|
||||
return nil, fmt.Errorf("parsing ledger: %w", err)
|
||||
}
|
||||
for i, t := range l.Tokens {
|
||||
if strings.TrimSpace(t.Token) == "" {
|
||||
return nil, fmt.Errorf("ledger tokens[%d] has an empty token", i)
|
||||
}
|
||||
}
|
||||
for i, p := range l.Pairs {
|
||||
if strings.TrimSpace(p.Parent) == "" || strings.TrimSpace(p.Variant) == "" {
|
||||
return nil, fmt.Errorf("ledger pairs[%d] needs both parent and variant", i)
|
||||
}
|
||||
}
|
||||
return l, nil
|
||||
}
|
||||
|
||||
// Suppression is a ledger hit: why a candidate was not proposed, in words a
|
||||
// reviewer can check against the ledger file.
|
||||
type Suppression struct {
|
||||
A string
|
||||
B string
|
||||
Reason string
|
||||
}
|
||||
|
||||
func (s Suppression) String() string {
|
||||
return fmt.Sprintf("%s + %s: %s", s.A, s.B, s.Reason)
|
||||
}
|
||||
|
||||
// Suppresses reports whether the ledger has already declined pairing these two
|
||||
// entries, and why.
|
||||
//
|
||||
// The token rule is applied to the segments the two names do not share. Two
|
||||
// builds of the same weights differ only in quantization markers, so any
|
||||
// ledgered token showing up in that difference is by construction a claim that
|
||||
// the entries are different models.
|
||||
func (l *Ledger) Suppresses(a, b string) (Suppression, bool) {
|
||||
la, lb := strings.ToLower(a), strings.ToLower(b)
|
||||
for _, p := range l.Pairs {
|
||||
lp, lv := strings.ToLower(p.Parent), strings.ToLower(p.Variant)
|
||||
if (lp == la && lv == lb) || (lp == lb && lv == la) {
|
||||
return Suppression{A: a, B: b, Reason: p.Reason}, true
|
||||
}
|
||||
}
|
||||
for _, g := range l.Groups {
|
||||
var seenA, seenB bool
|
||||
for _, m := range g.Members {
|
||||
lm := strings.ToLower(m)
|
||||
if lm == la {
|
||||
seenA = true
|
||||
}
|
||||
if lm == lb {
|
||||
seenB = true
|
||||
}
|
||||
}
|
||||
if seenA && seenB {
|
||||
return Suppression{A: a, B: b, Reason: g.Reason}, true
|
||||
}
|
||||
}
|
||||
diff := differingSegments(la, lb)
|
||||
for _, t := range l.Tokens {
|
||||
token := strings.ToLower(strings.TrimSpace(t.Token))
|
||||
if _, ok := diff[token]; ok {
|
||||
reason := t.Reason
|
||||
if reason == "" {
|
||||
reason = fmt.Sprintf("names differ by %q", token)
|
||||
}
|
||||
return Suppression{A: a, B: b, Reason: fmt.Sprintf("%s (token %q)", reason, token)}, true
|
||||
}
|
||||
}
|
||||
return Suppression{}, false
|
||||
}
|
||||
|
||||
// segments splits a name into the atoms the token rules are written against.
|
||||
func segments(name string) []string {
|
||||
fields := strings.FieldsFunc(strings.ToLower(name), func(r rune) bool {
|
||||
return r == '-' || r == '_' || r == '.' || r == ':' || r == '/'
|
||||
})
|
||||
return fields
|
||||
}
|
||||
|
||||
// differingSegments returns the set of segments present in exactly one of the
|
||||
// two names.
|
||||
func differingSegments(a, b string) map[string]struct{} {
|
||||
setA := map[string]int{}
|
||||
for _, s := range segments(a) {
|
||||
setA[s]++
|
||||
}
|
||||
setB := map[string]int{}
|
||||
for _, s := range segments(b) {
|
||||
setB[s]++
|
||||
}
|
||||
diff := map[string]struct{}{}
|
||||
for s := range setA {
|
||||
if setB[s] == 0 {
|
||||
diff[s] = struct{}{}
|
||||
}
|
||||
}
|
||||
for s := range setB {
|
||||
if setA[s] == 0 {
|
||||
diff[s] = struct{}{}
|
||||
}
|
||||
}
|
||||
return diff
|
||||
}
|
||||
|
||||
// SortedSuppressions gives the ledger's effect on one run in a stable order, so
|
||||
// the pull request body reads the same way for the same gallery.
|
||||
func SortedSuppressions(in []Suppression) []Suppression {
|
||||
out := append([]Suppression(nil), in...)
|
||||
sort.Slice(out, func(i, j int) bool {
|
||||
if out[i].A != out[j].A {
|
||||
return out[i].A < out[j].A
|
||||
}
|
||||
return out[i].B < out[j].B
|
||||
})
|
||||
return out
|
||||
}
|
||||
65
.github/ci/variantproposals/main.go
vendored
Normal file
65
.github/ci/variantproposals/main.go
vendored
Normal file
@@ -0,0 +1,65 @@
|
||||
// Command variant-proposals looks for gallery entries that are alternative
|
||||
// builds of the same weights but are not grouped under one another, and writes
|
||||
// a proposal for a human to accept or reject.
|
||||
//
|
||||
// It never decides. Grouping has gone wrong repeatedly in both directions, so
|
||||
// the job's value is catching drift and surfacing candidates with their
|
||||
// evidence, not automating the call. The scheduled workflow feeds its output to
|
||||
// a pull request in the same shape as .github/checksum_checker.sh.
|
||||
package main
|
||||
|
||||
import (
|
||||
"flag"
|
||||
"fmt"
|
||||
"os"
|
||||
"strings"
|
||||
)
|
||||
|
||||
func main() {
|
||||
index := flag.String("index", "gallery/index.yaml", "path to the gallery index")
|
||||
ledger := flag.String("ledger", "gallery/variant-exclusions.yaml", "path to the rejection ledger")
|
||||
bodyOut := flag.String("body-out", "", "write the pull request body here")
|
||||
apply := flag.Bool("apply", false, "write the proposed groupings back into the index")
|
||||
flag.Parse()
|
||||
|
||||
if err := run(*index, *ledger, *bodyOut, *apply); err != nil {
|
||||
fmt.Fprintln(os.Stderr, "variant-proposals:", err)
|
||||
os.Exit(1)
|
||||
}
|
||||
}
|
||||
|
||||
func run(indexPath, ledgerPath, bodyOut string, apply bool) error {
|
||||
ix, err := LoadIndex(indexPath)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
ledger, err := LoadLedger(ledgerPath)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
result := Propose(ix, ledger)
|
||||
fmt.Print(RenderSummary(result))
|
||||
|
||||
if !result.HasProposals() {
|
||||
// An empty pull request every night is how a proposal job gets muted.
|
||||
fmt.Println("nothing to propose")
|
||||
return nil
|
||||
}
|
||||
|
||||
if bodyOut != "" {
|
||||
if err := os.WriteFile(bodyOut, []byte(RenderBody(result, ledgerPath)), 0o644); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
|
||||
if !apply {
|
||||
return nil
|
||||
}
|
||||
|
||||
lines, err := ApplyFamilies(ix, result.Families)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return os.WriteFile(indexPath, []byte(strings.Join(lines, "\n")), 0o644)
|
||||
}
|
||||
615
.github/ci/variantproposals/propose.go
vendored
Normal file
615
.github/ci/variantproposals/propose.go
vendored
Normal file
@@ -0,0 +1,615 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"regexp"
|
||||
"sort"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// Signal names the grouping heuristic that linked two entries.
|
||||
type Signal string
|
||||
|
||||
const (
|
||||
// SignalName is "same name once quantization markers are stripped".
|
||||
SignalName Signal = "name-modulo-quant"
|
||||
// SignalConfigSuffix is the ":" convention, foo:q8_0 as a build of foo.
|
||||
SignalConfigSuffix Signal = "config-suffix"
|
||||
// SignalWeightFile is "same primary weight filename once quantization
|
||||
// markers are stripped", auxiliary files excluded.
|
||||
SignalWeightFile Signal = "weight-filename"
|
||||
)
|
||||
|
||||
// Evidence is what a reviewer needs in order to agree or disagree without
|
||||
// opening HuggingFace: what the two entries share, and what differs.
|
||||
type Evidence struct {
|
||||
Signals []Signal
|
||||
SharedStem string
|
||||
SharedFile string
|
||||
SharedRepo string
|
||||
QuantTokens []string
|
||||
}
|
||||
|
||||
// Proposal is one variant target offered to one parent.
|
||||
type Proposal struct {
|
||||
Variant string
|
||||
Evidence Evidence
|
||||
}
|
||||
|
||||
// Family is a complete proposal: one parent gaining one or more variants.
|
||||
type Family struct {
|
||||
Parent string
|
||||
Proposals []Proposal
|
||||
}
|
||||
|
||||
// Refusal is a family the heuristics found but the rules would not let through.
|
||||
// Refusals are reported rather than dropped: a candidate the job keeps refusing
|
||||
// is either a rule worth revisiting or a gallery bug worth fixing.
|
||||
type Refusal struct {
|
||||
Members []string
|
||||
Reason string
|
||||
}
|
||||
|
||||
// Result is one run of the proposer.
|
||||
type Result struct {
|
||||
Families []Family
|
||||
Refusals []Refusal
|
||||
Suppressed []Suppression
|
||||
AliasSkipped []Suppression
|
||||
}
|
||||
|
||||
// HasProposals reports whether the run found anything to open a pull request
|
||||
// about. A job that opens an empty pull request every night is a job people
|
||||
// filter out of their inbox.
|
||||
func (r *Result) HasProposals() bool {
|
||||
return len(r.Families) > 0
|
||||
}
|
||||
|
||||
// sizeToken matches a parameter-count marker: 8b, 1.7b, a3b for an active
|
||||
// expert count, e2b for the Gemma effective sizes, 8x7b for a mixture.
|
||||
//
|
||||
// This is a structural rule rather than a ledger entry because it is about the
|
||||
// shape of the token, not about any one model. Different parameter sizes were
|
||||
// mis-grouped by an earlier sweep and the failure is systematic.
|
||||
var sizeToken = regexp.MustCompile(`^(?:[0-9]+(?:\.[0-9]+)?[bm]|[ae][0-9]+(?:\.[0-9]+)?b|[0-9]+x[0-9]+(?:\.[0-9]+)?b)$`)
|
||||
|
||||
func differsByParameterSize(a, b string) (string, bool) {
|
||||
for seg := range differingSegments(a, b) {
|
||||
if sizeToken.MatchString(seg) {
|
||||
return seg, true
|
||||
}
|
||||
}
|
||||
return "", false
|
||||
}
|
||||
|
||||
// genericFileStem lists weight filenames too generic to be evidence of
|
||||
// anything. Two entries both shipping "model.safetensors" share a convention,
|
||||
// not a set of weights.
|
||||
var genericFileStem = map[string]struct{}{
|
||||
"model": {}, "weights": {}, "pytorch_model": {}, "diffusion_pytorch_model": {},
|
||||
"consolidated": {}, "ggml-model": {}, "model-00001-of-00002": {},
|
||||
}
|
||||
|
||||
// minFileStemLength keeps short, collision-prone filename stems from linking
|
||||
// unrelated entries.
|
||||
const minFileStemLength = 6
|
||||
|
||||
type pair struct {
|
||||
a, b int
|
||||
evidence Evidence
|
||||
}
|
||||
|
||||
// Propose runs the grouping heuristics over a gallery index and returns what it
|
||||
// would offer a human, what it refused, and what the ledger silenced.
|
||||
//
|
||||
// Nothing here touches the network or git, and the index is not modified.
|
||||
func Propose(ix *Index, ledger *Ledger) *Result {
|
||||
if ledger == nil {
|
||||
ledger = &Ledger{}
|
||||
}
|
||||
result := &Result{}
|
||||
|
||||
byName, dupes := ix.ByName()
|
||||
|
||||
// Existing relationships. A target already claimed must not be claimed
|
||||
// again, and two entries already in one family need no proposal.
|
||||
claimedBy := map[string]string{}
|
||||
familyOf := map[string]string{}
|
||||
for _, e := range ix.Entries {
|
||||
if !e.HasVariants() {
|
||||
continue
|
||||
}
|
||||
familyOf[strings.ToLower(e.Name)] = strings.ToLower(e.Name)
|
||||
for _, v := range e.Variants {
|
||||
target := strings.ToLower(v.Model)
|
||||
if _, taken := claimedBy[target]; !taken {
|
||||
claimedBy[target] = strings.ToLower(e.Name)
|
||||
}
|
||||
familyOf[target] = strings.ToLower(e.Name)
|
||||
}
|
||||
}
|
||||
|
||||
candidates := map[[2]int]*Evidence{}
|
||||
|
||||
addPair := func(i, j int, sig Signal, apply func(*Evidence)) {
|
||||
if i == j {
|
||||
return
|
||||
}
|
||||
if i > j {
|
||||
i, j = j, i
|
||||
}
|
||||
key := [2]int{i, j}
|
||||
ev, ok := candidates[key]
|
||||
if !ok {
|
||||
ev = &Evidence{}
|
||||
candidates[key] = ev
|
||||
}
|
||||
for _, s := range ev.Signals {
|
||||
if s == sig {
|
||||
apply(ev)
|
||||
return
|
||||
}
|
||||
}
|
||||
ev.Signals = append(ev.Signals, sig)
|
||||
apply(ev)
|
||||
}
|
||||
|
||||
// Signal 1 and 2: entries sharing a name stem.
|
||||
byStem := map[string][]int{}
|
||||
for _, e := range ix.Entries {
|
||||
if e.Name == "" {
|
||||
continue
|
||||
}
|
||||
byStem[NameStem(e.Name)] = append(byStem[NameStem(e.Name)], e.Index)
|
||||
}
|
||||
for stem, members := range byStem {
|
||||
if len(members) < 2 {
|
||||
continue
|
||||
}
|
||||
for i := 0; i < len(members); i++ {
|
||||
for j := i + 1; j < len(members); j++ {
|
||||
a, b := ix.Entries[members[i]], ix.Entries[members[j]]
|
||||
sig := SignalName
|
||||
if HasConfigSuffix(a.Name) || HasConfigSuffix(b.Name) {
|
||||
sig = SignalConfigSuffix
|
||||
}
|
||||
// The bare parent carries no marker in its name, so the
|
||||
// evidence would read "differs by q8_0" and say nothing about
|
||||
// what the parent is. The weight filenames fill that in.
|
||||
fa, _ := a.PrimaryWeightFile()
|
||||
fb, _ := b.PrimaryWeightFile()
|
||||
addPair(members[i], members[j], sig, func(ev *Evidence) {
|
||||
ev.SharedStem = stem
|
||||
ev.QuantTokens = quantDifference(a.Name, b.Name, fa, fb)
|
||||
})
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Signal 3: entries whose own weight file is the same file at a different
|
||||
// quantization. Auxiliary files never take part.
|
||||
byFile := map[string][]int{}
|
||||
for _, e := range ix.Entries {
|
||||
primary, ok := e.PrimaryWeightFile()
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
stem := FileStem(primary)
|
||||
if len(stem) < minFileStemLength {
|
||||
continue
|
||||
}
|
||||
if _, generic := genericFileStem[stem]; generic {
|
||||
continue
|
||||
}
|
||||
byFile[stem] = append(byFile[stem], e.Index)
|
||||
}
|
||||
for stem, members := range byFile {
|
||||
if len(members) < 2 {
|
||||
continue
|
||||
}
|
||||
for i := 0; i < len(members); i++ {
|
||||
for j := i + 1; j < len(members); j++ {
|
||||
a, b := ix.Entries[members[i]], ix.Entries[members[j]]
|
||||
// The filename alone is not evidence. Publishers reuse the
|
||||
// upstream filename for finetunes and for models that merely
|
||||
// embed the base weights: bert-embeddings, an ultravox audio
|
||||
// model and a roleplay finetune all ship a file called
|
||||
// llama-3.2-1b-instruct-q4_k_m.gguf. Requiring the same
|
||||
// upstream repository turns the signal back into what it
|
||||
// claims to be, one repo publishing one file at two
|
||||
// quantizations. Two repos holding the same weights is a fact
|
||||
// no filename proves, so it stays a human call.
|
||||
repo := a.SourceRepo()
|
||||
if repo == "" || repo != b.SourceRepo() {
|
||||
continue
|
||||
}
|
||||
fa, _ := a.PrimaryWeightFile()
|
||||
fb, _ := b.PrimaryWeightFile()
|
||||
addPair(members[i], members[j], SignalWeightFile, func(ev *Evidence) {
|
||||
ev.SharedFile = stem
|
||||
ev.SharedRepo = repo
|
||||
if len(ev.QuantTokens) == 0 {
|
||||
ev.QuantTokens = quantDifference(fa, fb)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Filter candidates. Everything dropped here is dropped for a reason a
|
||||
// reviewer can read back off the ledger or the rules.
|
||||
var kept []pair
|
||||
for key, ev := range candidates {
|
||||
a, b := ix.Entries[key[0]], ix.Entries[key[1]]
|
||||
la, lb := strings.ToLower(a.Name), strings.ToLower(b.Name)
|
||||
if la == lb {
|
||||
continue
|
||||
}
|
||||
if dupes[la] > 0 || dupes[lb] > 0 {
|
||||
result.Refusals = append(result.Refusals, Refusal{
|
||||
Members: []string{a.Name, b.Name},
|
||||
Reason: "one of these names appears more than once in the gallery, so a variant reference to it is ambiguous",
|
||||
})
|
||||
continue
|
||||
}
|
||||
if fa, fb := familyOf[la], familyOf[lb]; fa != "" && fa == fb {
|
||||
continue
|
||||
}
|
||||
if seg, differs := differsByParameterSize(la, lb); differs {
|
||||
result.Suppressed = append(result.Suppressed, Suppression{
|
||||
A: a.Name, B: b.Name, Reason: fmt.Sprintf("different parameter sizes (segment %q)", seg),
|
||||
})
|
||||
continue
|
||||
}
|
||||
if s, ok := ledger.Suppresses(a.Name, b.Name); ok {
|
||||
result.Suppressed = append(result.Suppressed, s)
|
||||
continue
|
||||
}
|
||||
if SameInstallPayload(a, b) {
|
||||
result.AliasSkipped = append(result.AliasSkipped, Suppression{
|
||||
A: a.Name, B: b.Name,
|
||||
Reason: "identical install payload; these are aliases of one build, not alternative builds",
|
||||
})
|
||||
continue
|
||||
}
|
||||
kept = append(kept, pair{a: key[0], b: key[1], evidence: *ev})
|
||||
}
|
||||
|
||||
sort.Slice(kept, func(i, j int) bool {
|
||||
if kept[i].a != kept[j].a {
|
||||
return kept[i].a < kept[j].a
|
||||
}
|
||||
return kept[i].b < kept[j].b
|
||||
})
|
||||
|
||||
// Components. A pair from either signal joins the same family, so a chain
|
||||
// of alternative builds discovered by different signals stays one family
|
||||
// rather than two overlapping ones that would double claim a target.
|
||||
parent := map[int]int{}
|
||||
var find func(int) int
|
||||
find = func(x int) int {
|
||||
if p, ok := parent[x]; ok && p != x {
|
||||
parent[x] = find(p)
|
||||
return parent[x]
|
||||
}
|
||||
if _, ok := parent[x]; !ok {
|
||||
parent[x] = x
|
||||
}
|
||||
return parent[x]
|
||||
}
|
||||
union := func(x, y int) {
|
||||
rx, ry := find(x), find(y)
|
||||
if rx != ry {
|
||||
parent[ry] = rx
|
||||
}
|
||||
}
|
||||
evidenceFor := map[[2]int]Evidence{}
|
||||
for _, p := range kept {
|
||||
union(p.a, p.b)
|
||||
evidenceFor[[2]int{p.a, p.b}] = p.evidence
|
||||
}
|
||||
|
||||
components := map[int][]int{}
|
||||
for _, p := range kept {
|
||||
for _, m := range []int{p.a, p.b} {
|
||||
root := find(m)
|
||||
if !contains(components[root], m) {
|
||||
components[root] = append(components[root], m)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
roots := make([]int, 0, len(components))
|
||||
for r := range components {
|
||||
roots = append(roots, r)
|
||||
}
|
||||
sort.Ints(roots)
|
||||
|
||||
proposedTargets := map[string]string{}
|
||||
for _, root := range roots {
|
||||
members := components[root]
|
||||
sort.Ints(members)
|
||||
family, refusal := buildFamily(ix, members, evidenceFor, claimedBy, proposedTargets, byName)
|
||||
if refusal != nil {
|
||||
result.Refusals = append(result.Refusals, *refusal)
|
||||
continue
|
||||
}
|
||||
if family == nil {
|
||||
continue
|
||||
}
|
||||
for _, p := range family.Proposals {
|
||||
proposedTargets[strings.ToLower(p.Variant)] = family.Parent
|
||||
}
|
||||
result.Families = append(result.Families, *family)
|
||||
}
|
||||
|
||||
sort.Slice(result.Families, func(i, j int) bool { return result.Families[i].Parent < result.Families[j].Parent })
|
||||
result.Suppressed = SortedSuppressions(result.Suppressed)
|
||||
result.AliasSkipped = SortedSuppressions(result.AliasSkipped)
|
||||
result.Refusals = dedupeRefusals(result.Refusals)
|
||||
return result
|
||||
}
|
||||
|
||||
// dedupeRefusals collapses the same refusal reached from both orderings of a
|
||||
// pair, and sorts what is left. A reviewer reading the same complaint twice
|
||||
// learns to skim the section.
|
||||
func dedupeRefusals(in []Refusal) []Refusal {
|
||||
seen := map[string]struct{}{}
|
||||
var out []Refusal
|
||||
for _, r := range in {
|
||||
members := append([]string(nil), r.Members...)
|
||||
sort.Strings(members)
|
||||
key := strings.Join(members, "\x00") + "\x00" + r.Reason
|
||||
if _, dup := seen[key]; dup {
|
||||
continue
|
||||
}
|
||||
seen[key] = struct{}{}
|
||||
out = append(out, r)
|
||||
}
|
||||
sort.Slice(out, func(i, j int) bool {
|
||||
if a, b := strings.Join(out[i].Members, ","), strings.Join(out[j].Members, ","); a != b {
|
||||
return a < b
|
||||
}
|
||||
return out[i].Reason < out[j].Reason
|
||||
})
|
||||
return out
|
||||
}
|
||||
|
||||
func contains(xs []int, x int) bool {
|
||||
for _, v := range xs {
|
||||
if v == x {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// buildFamily turns a connected component into a proposal, or refuses it.
|
||||
func buildFamily(ix *Index, members []int, evidenceFor map[[2]int]Evidence, claimedBy map[string]string, proposedTargets map[string]string, byName map[string]*GalleryEntry) (*Family, *Refusal) {
|
||||
names := make([]string, 0, len(members))
|
||||
for _, m := range members {
|
||||
names = append(names, ix.Entries[m].Name)
|
||||
}
|
||||
|
||||
parentIdx, err := selectParent(ix, members)
|
||||
if err != nil {
|
||||
return nil, &Refusal{Members: names, Reason: err.Error()}
|
||||
}
|
||||
parentEntry := ix.Entries[parentIdx]
|
||||
parentName := strings.ToLower(parentEntry.Name)
|
||||
|
||||
// A parent that is itself somebody's variant would create a chain, which
|
||||
// the gallery's own resolution refuses to install.
|
||||
if owner, claimed := claimedBy[parentName]; claimed {
|
||||
return nil, &Refusal{Members: names, Reason: fmt.Sprintf("the natural parent %q is already a variant of %q; proposing it as a parent would nest variants", parentEntry.Name, owner)}
|
||||
}
|
||||
if owner, claimed := proposedTargets[parentName]; claimed {
|
||||
return nil, &Refusal{Members: names, Reason: fmt.Sprintf("the natural parent %q is already proposed as a variant of %q; proposing it as a parent would nest variants", parentEntry.Name, owner)}
|
||||
}
|
||||
|
||||
// Adding a variants key to an anchor is inherited by every entry that
|
||||
// merges it, silently grouping models nobody proposed. Handling that means
|
||||
// editing each merging child too, which is a larger change than this job
|
||||
// should make unsupervised, so it refuses and hands the reviewer the list.
|
||||
if parentEntry.AnchorName != "" {
|
||||
children := ix.MergeChildren(parentEntry.AnchorName)
|
||||
if len(children) > 0 {
|
||||
childNames := make([]string, 0, len(children))
|
||||
for _, c := range children {
|
||||
childNames = append(childNames, c.Name)
|
||||
}
|
||||
return nil, &Refusal{
|
||||
Members: names,
|
||||
Reason: fmt.Sprintf("the parent %q defines YAML anchor &%s, and a variants key added there is inherited by the %d entries that merge it (%s). Grouping this family by hand also means adding an explicit `variants: []` to each of those entries",
|
||||
parentEntry.Name, parentEntry.AnchorName, len(children), strings.Join(childNames, ", ")),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
existing := map[string]struct{}{}
|
||||
for _, v := range parentEntry.Variants {
|
||||
existing[strings.ToLower(v.Model)] = struct{}{}
|
||||
}
|
||||
|
||||
family := &Family{Parent: parentEntry.Name}
|
||||
for _, m := range members {
|
||||
if m == parentIdx {
|
||||
continue
|
||||
}
|
||||
target := ix.Entries[m]
|
||||
lower := strings.ToLower(target.Name)
|
||||
if _, already := existing[lower]; already {
|
||||
continue
|
||||
}
|
||||
if target.HasVariants() {
|
||||
return nil, &Refusal{Members: names, Reason: fmt.Sprintf("%q already offers variants of its own, so it cannot itself be a variant target", target.Name)}
|
||||
}
|
||||
if !target.Installable() {
|
||||
return nil, &Refusal{Members: names, Reason: fmt.Sprintf("%q has no url, config_file, overrides or files, so it is not independently installable", target.Name)}
|
||||
}
|
||||
if owner, claimed := claimedBy[lower]; claimed && owner != parentName {
|
||||
return nil, &Refusal{Members: names, Reason: fmt.Sprintf("%q is already a variant of %q; a target claimed by two parents is not something the gallery resolves predictably", target.Name, owner)}
|
||||
}
|
||||
if owner, claimed := proposedTargets[lower]; claimed && owner != parentEntry.Name {
|
||||
return nil, &Refusal{Members: names, Reason: fmt.Sprintf("%q is already proposed as a variant of %q in this same run", target.Name, owner)}
|
||||
}
|
||||
family.Proposals = append(family.Proposals, Proposal{
|
||||
Variant: target.Name,
|
||||
Evidence: lookupEvidence(evidenceFor, parentIdx, m),
|
||||
})
|
||||
}
|
||||
|
||||
if len(family.Proposals) == 0 {
|
||||
return nil, nil
|
||||
}
|
||||
sort.Slice(family.Proposals, func(i, j int) bool { return family.Proposals[i].Variant < family.Proposals[j].Variant })
|
||||
return family, nil
|
||||
}
|
||||
|
||||
func lookupEvidence(evidenceFor map[[2]int]Evidence, a, b int) Evidence {
|
||||
if a > b {
|
||||
a, b = b, a
|
||||
}
|
||||
if ev, ok := evidenceFor[[2]int{a, b}]; ok {
|
||||
return ev
|
||||
}
|
||||
// The two entries reached the same family through a third one. Say so
|
||||
// rather than inventing evidence that was never observed for this pair.
|
||||
return Evidence{Signals: []Signal{SignalName}}
|
||||
}
|
||||
|
||||
// selectParent picks the entry the others should hang off.
|
||||
//
|
||||
// The bare name wins when there is one: it is the name a user types and the one
|
||||
// documentation links to. Otherwise the smallest build wins, judged by the
|
||||
// quantization token in the entry's own weight filename, so the default install
|
||||
// is the one most hosts can actually run.
|
||||
func selectParent(ix *Index, members []int) (int, error) {
|
||||
// The family's own stem: the one the most members reduce to, shortest name
|
||||
// breaking a tie. An entry named exactly that is the bare entry.
|
||||
stemCount := map[string]int{}
|
||||
for _, m := range members {
|
||||
stemCount[NameStem(ix.Entries[m].Name)]++
|
||||
}
|
||||
// Only a stem two or more members reduce to is the family's own stem. A
|
||||
// stem reached by exactly one member is just that member's name, and
|
||||
// treating it as the family stem would crown whichever name happens to be
|
||||
// shortest rather than whichever build is the base one.
|
||||
familyStem := ""
|
||||
for stem, n := range stemCount {
|
||||
if n < 2 {
|
||||
continue
|
||||
}
|
||||
if familyStem == "" || n > stemCount[familyStem] ||
|
||||
(n == stemCount[familyStem] && len(stem) < len(familyStem)) ||
|
||||
(n == stemCount[familyStem] && len(stem) == len(familyStem) && stem < familyStem) {
|
||||
familyStem = stem
|
||||
}
|
||||
}
|
||||
|
||||
var bare []int
|
||||
for _, m := range members {
|
||||
e := ix.Entries[m]
|
||||
if HasConfigSuffix(e.Name) {
|
||||
continue
|
||||
}
|
||||
if strings.ToLower(e.Name) == familyStem {
|
||||
bare = append(bare, m)
|
||||
}
|
||||
}
|
||||
if len(bare) == 1 {
|
||||
return bare[0], nil
|
||||
}
|
||||
if len(bare) > 1 {
|
||||
names := make([]string, 0, len(bare))
|
||||
for _, m := range bare {
|
||||
names = append(names, ix.Entries[m].Name)
|
||||
}
|
||||
return 0, fmt.Errorf("more than one entry is named exactly %q (%s), so which one is the base build is a judgement this job will not make", familyStem, strings.Join(names, ", "))
|
||||
}
|
||||
|
||||
// No shared stem to be named after. An entry whose name every other member
|
||||
// extends is still recognisably the base one, and this is the only handle
|
||||
// left for families whose weights carry no readable quantization token at
|
||||
// all, such as the ONNX builds.
|
||||
if prefix, ok := uniquePrefixMember(ix, members); ok {
|
||||
return prefix, nil
|
||||
}
|
||||
|
||||
best := -1
|
||||
bestWidth := 1 << 20
|
||||
for _, m := range members {
|
||||
e := ix.Entries[m]
|
||||
width := unknownWidth
|
||||
if primary, ok := e.PrimaryWeightFile(); ok {
|
||||
width = BuildWidth(primary)
|
||||
}
|
||||
// Members are visited in gallery order, so a strict comparison leaves
|
||||
// the earliest entry holding a tie and the choice is deterministic.
|
||||
if width < bestWidth {
|
||||
best, bestWidth = m, width
|
||||
}
|
||||
}
|
||||
if best < 0 {
|
||||
return 0, fmt.Errorf("no member could be identified as the smallest build")
|
||||
}
|
||||
if bestWidth == unknownWidth {
|
||||
names := make([]string, 0, len(members))
|
||||
for _, m := range members {
|
||||
names = append(names, ix.Entries[m].Name)
|
||||
}
|
||||
return 0, fmt.Errorf("no member declares a weight file whose quantization can be read (%s), so the smallest build cannot be identified", strings.Join(names, ", "))
|
||||
}
|
||||
return best, nil
|
||||
}
|
||||
|
||||
// uniquePrefixMember reports the single member whose name every other member's
|
||||
// name starts with, if there is exactly one.
|
||||
func uniquePrefixMember(ix *Index, members []int) (int, bool) {
|
||||
found := -1
|
||||
for _, m := range members {
|
||||
name := strings.ToLower(ix.Entries[m].Name)
|
||||
isPrefix := true
|
||||
for _, other := range members {
|
||||
if other == m {
|
||||
continue
|
||||
}
|
||||
if !strings.HasPrefix(strings.ToLower(ix.Entries[other].Name), name) {
|
||||
isPrefix = false
|
||||
break
|
||||
}
|
||||
}
|
||||
if !isPrefix {
|
||||
continue
|
||||
}
|
||||
if found >= 0 {
|
||||
return 0, false
|
||||
}
|
||||
found = m
|
||||
}
|
||||
return found, found >= 0
|
||||
}
|
||||
|
||||
// quantDifference lists the quantization tokens that tell two names apart. It
|
||||
// is the compact form of the evidence: "these differ only by q4_k_m vs q8_0".
|
||||
func quantDifference(names ...string) []string {
|
||||
var out []string
|
||||
seen := map[string]struct{}{}
|
||||
for _, name := range names {
|
||||
// Filenames arrive here too, so the extension goes first and "/" counts
|
||||
// as a separator. "_" deliberately does not: it holds "q4_k_m" together.
|
||||
trimmed := weightExtension.ReplaceAllString(name, "")
|
||||
for _, seg := range strings.FieldsFunc(strings.ToLower(trimmed), func(r rune) bool { return r == '-' || r == '/' }) {
|
||||
if !IsQuantToken(seg) {
|
||||
continue
|
||||
}
|
||||
if _, ok := seen[seg]; ok {
|
||||
continue
|
||||
}
|
||||
seen[seg] = struct{}{}
|
||||
out = append(out, seg)
|
||||
}
|
||||
}
|
||||
sort.Strings(out)
|
||||
return out
|
||||
}
|
||||
429
.github/ci/variantproposals/propose_test.go
vendored
Normal file
429
.github/ci/variantproposals/propose_test.go
vendored
Normal file
@@ -0,0 +1,429 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
)
|
||||
|
||||
// entryYAML writes one gallery entry with a single weight file, which is the
|
||||
// shape almost every real entry has. Specs that need something else write the
|
||||
// YAML out by hand.
|
||||
func entryYAML(name, repo, filename, sha string) string {
|
||||
return fmt.Sprintf(`- name: %s
|
||||
url: github:mudler/LocalAI/gallery/virtual.yaml@master
|
||||
overrides:
|
||||
parameters:
|
||||
model: %s
|
||||
files:
|
||||
- filename: %s
|
||||
uri: huggingface://%s/%s
|
||||
sha256: %s
|
||||
`, name, filename, filename, repo, filename, sha)
|
||||
}
|
||||
|
||||
func indexOf(entries ...string) *Index {
|
||||
ix, err := ParseIndex(strings.Join(entries, ""))
|
||||
ExpectWithOffset(1, err).ToNot(HaveOccurred())
|
||||
return ix
|
||||
}
|
||||
|
||||
// familyNames flattens a result into "parent <- variant, variant" strings, the
|
||||
// form the specs assert against.
|
||||
func familyNames(r *Result) []string {
|
||||
out := make([]string, 0, len(r.Families))
|
||||
for _, f := range r.Families {
|
||||
names := make([]string, 0, len(f.Proposals))
|
||||
for _, p := range f.Proposals {
|
||||
names = append(names, p.Variant)
|
||||
}
|
||||
out = append(out, f.Parent+" <- "+strings.Join(names, ", "))
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func refusalReasons(r *Result) string {
|
||||
var b strings.Builder
|
||||
for _, ref := range r.Refusals {
|
||||
b.WriteString(strings.Join(ref.Members, " + ") + ": " + ref.Reason + "\n")
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
|
||||
func suppressionReasons(r *Result) string {
|
||||
var b strings.Builder
|
||||
for _, s := range r.Suppressed {
|
||||
b.WriteString(s.String() + "\n")
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
|
||||
var _ = Describe("Propose", func() {
|
||||
Describe("the grouping signals", func() {
|
||||
It("groups entries whose names differ only by a quantization marker", func() {
|
||||
ix := indexOf(
|
||||
entryYAML("foo-model", "acme/foo-GGUF", "foo-model-Q4_K_M.gguf", "aa"),
|
||||
entryYAML("foo-model-q8_0", "acme/foo-GGUF", "foo-model-Q8_0.gguf", "bb"),
|
||||
)
|
||||
r := Propose(ix, nil)
|
||||
Expect(familyNames(r)).To(ConsistOf("foo-model <- foo-model-q8_0"))
|
||||
Expect(r.Families[0].Proposals[0].Evidence.Signals).To(ContainElement(SignalName))
|
||||
Expect(r.Families[0].Proposals[0].Evidence.SharedStem).To(Equal("foo-model"))
|
||||
Expect(r.Families[0].Proposals[0].Evidence.QuantTokens).To(ContainElements("q4_k_m", "q8_0"))
|
||||
})
|
||||
|
||||
It("groups entries that use the colon config-suffix convention", func() {
|
||||
ix := indexOf(
|
||||
entryYAML("bar-model", "acme/bar-GGUF", "bar-model-Q4_K_M.gguf", "aa"),
|
||||
entryYAML("bar-model:grammar-functioncall", "acme/bar-GGUF", "bar-model-Q4_K_M-grammar.gguf", "bb"),
|
||||
)
|
||||
r := Propose(ix, nil)
|
||||
Expect(familyNames(r)).To(ConsistOf("bar-model <- bar-model:grammar-functioncall"))
|
||||
Expect(r.Families[0].Proposals[0].Evidence.Signals).To(ContainElement(SignalConfigSuffix))
|
||||
})
|
||||
|
||||
It("groups entries whose own weight file is the same file at another quantization", func() {
|
||||
// The names share no stem, so only the filename signal can link
|
||||
// these two.
|
||||
ix := indexOf(
|
||||
entryYAML("omni-cpp", "Serveurperso/Omni-GGUF", "omnivoice-base-Q8_0.gguf", "aa"),
|
||||
entryYAML("omni-cpp-hq", "Serveurperso/Omni-GGUF", "omnivoice-base-BF16.gguf", "bb"),
|
||||
)
|
||||
r := Propose(ix, nil)
|
||||
Expect(familyNames(r)).To(ConsistOf("omni-cpp <- omni-cpp-hq"))
|
||||
ev := r.Families[0].Proposals[0].Evidence
|
||||
Expect(ev.Signals).To(ConsistOf(SignalWeightFile))
|
||||
Expect(ev.SharedFile).To(Equal("omnivoice-base"))
|
||||
Expect(ev.SharedRepo).To(Equal("serveurperso/omni-gguf"))
|
||||
})
|
||||
|
||||
It("does not let a shared auxiliary file link unrelated models", func() {
|
||||
// Both entries ship the same text encoder. That is a packaging
|
||||
// convention, not evidence of shared weights: this is how an
|
||||
// earlier sweep linked four wan-2.1 entries to each other.
|
||||
ix := indexOf(`- name: wan-2.1-t2v
|
||||
url: u
|
||||
files:
|
||||
- filename: wan-2.1-t2v-Q4_K_M.gguf
|
||||
uri: huggingface://acme/wan/wan-2.1-t2v-Q4_K_M.gguf
|
||||
sha256: aa
|
||||
- filename: umt5-xxl-encoder-Q8_0.gguf
|
||||
uri: huggingface://acme/wan/umt5-xxl-encoder-Q8_0.gguf
|
||||
sha256: cc
|
||||
`, `- name: z-image-turbo
|
||||
url: u
|
||||
files:
|
||||
- filename: z-image-turbo-Q4_K_M.gguf
|
||||
uri: huggingface://acme/wan/z-image-turbo-Q4_K_M.gguf
|
||||
sha256: bb
|
||||
- filename: umt5-xxl-encoder-Q8_0.gguf
|
||||
uri: huggingface://acme/wan/umt5-xxl-encoder-Q8_0.gguf
|
||||
sha256: cc
|
||||
`)
|
||||
r := Propose(ix, nil)
|
||||
Expect(familyNames(r)).To(BeEmpty())
|
||||
})
|
||||
|
||||
It("does not treat a shared filename in two different repos as evidence", func() {
|
||||
// A finetune republished under the base model's filename is the
|
||||
// most common way this signal misfires.
|
||||
ix := indexOf(
|
||||
entryYAML("llama-3.2-3b-instruct", "hugging-quants/Llama-3.2-3B-Instruct-GGUF", "llama-3.2-3b-instruct-q4_k_m.gguf", "aa"),
|
||||
entryYAML("llama-3.2-3b-shiro-roleplay", "someone/Shiro-GGUF", "Llama-3.2-3B-Instruct.Q8_0.gguf", "bb"),
|
||||
)
|
||||
r := Propose(ix, nil)
|
||||
Expect(familyNames(r)).To(BeEmpty())
|
||||
})
|
||||
})
|
||||
|
||||
Describe("what must never be proposed", func() {
|
||||
It("does not group different parameter sizes that share a prefix", func() {
|
||||
ix := indexOf(
|
||||
entryYAML("qwen3-tts-cpp-0.6b-base", "Serveurperso/Qwen3-TTS-GGUF", "qwen3-tts-talker-Q4_K_M.gguf", "aa"),
|
||||
entryYAML("qwen3-tts-cpp-1.7b-base", "Serveurperso/Qwen3-TTS-GGUF", "qwen3-tts-talker-Q8_0.gguf", "bb"),
|
||||
)
|
||||
r := Propose(ix, nil)
|
||||
Expect(familyNames(r)).To(BeEmpty())
|
||||
Expect(suppressionReasons(r)).To(ContainSubstring("different parameter sizes"))
|
||||
})
|
||||
|
||||
It("does not group the Gemma effective sizes", func() {
|
||||
ix := indexOf(
|
||||
entryYAML("gemma-4-e2b-it", "google/gemma-GGUF", "gemma-4-it-Q4_K_M.gguf", "aa"),
|
||||
entryYAML("gemma-4-e4b-it", "google/gemma-GGUF", "gemma-4-it-Q8_0.gguf", "bb"),
|
||||
)
|
||||
r := Propose(ix, nil)
|
||||
Expect(familyNames(r)).To(BeEmpty())
|
||||
Expect(suppressionReasons(r)).To(ContainSubstring("different parameter sizes"))
|
||||
})
|
||||
|
||||
It("does not group entries with a byte-identical install payload", func() {
|
||||
// whisper-1 exists so OpenAI-compatible clients can send that name.
|
||||
// Folding it under whisper-base would hide the name they send.
|
||||
payload := ` url: github:mudler/LocalAI/gallery/whisper-base.yaml@master
|
||||
overrides:
|
||||
parameters:
|
||||
model: ggml-whisper-base.bin
|
||||
files:
|
||||
- filename: ggml-whisper-base.bin
|
||||
uri: huggingface://ggerganov/whisper.cpp/ggml-base.bin
|
||||
sha256: aa
|
||||
`
|
||||
ix := indexOf("- name: whisper-base\n"+payload, "- name: whisper-1\n"+payload)
|
||||
r := Propose(ix, nil)
|
||||
Expect(familyNames(r)).To(BeEmpty())
|
||||
Expect(r.AliasSkipped).To(HaveLen(1))
|
||||
Expect(r.AliasSkipped[0].Reason).To(ContainSubstring("aliases"))
|
||||
})
|
||||
|
||||
DescribeTable("declines the categories the ledger records",
|
||||
func(nameA, nameB string, ledgerYAML string) {
|
||||
ix := indexOf(
|
||||
entryYAML(nameA, "acme/repo", "shared-weights-Q4_K_M.gguf", "aa"),
|
||||
entryYAML(nameB, "acme/repo", "shared-weights-Q8_0.gguf", "bb"),
|
||||
)
|
||||
ledger, err := ParseLedger([]byte(ledgerYAML))
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
|
||||
// Without the ledger these would be proposed, which is what
|
||||
// makes the ledger load bearing rather than decorative.
|
||||
Expect(familyNames(Propose(ix, nil))).ToNot(BeEmpty())
|
||||
|
||||
r := Propose(ix, ledger)
|
||||
Expect(familyNames(r)).To(BeEmpty())
|
||||
Expect(r.Suppressed).To(HaveLen(1))
|
||||
},
|
||||
Entry("a finetune", "base-model", "base-model-abliterated",
|
||||
"tokens:\n - {token: abliterated, reason: finetune}\n"),
|
||||
Entry("a distill", "base-model", "base-model-distilled",
|
||||
"tokens:\n - {token: distilled, reason: distilled}\n"),
|
||||
Entry("English-only versus multilingual ASR", "whisper-small", "whisper-small-en",
|
||||
"pairs:\n - {parent: whisper-small, variant: whisper-small-en, reason: English-only versus multilingual}\n"),
|
||||
Entry("two products sharing a prefix", "vibevoice-cpp", "vibevoice-cpp-asr",
|
||||
"pairs:\n - {parent: vibevoice-cpp, variant: vibevoice-cpp-asr, reason: different products}\n"),
|
||||
Entry("a per-language release", "kokoros-de", "kokoros-ja",
|
||||
"groups:\n - {members: [kokoros, kokoros-de, kokoros-ja], reason: different languages}\n"),
|
||||
)
|
||||
|
||||
It("reports the ledger's reason so its effect stays visible", func() {
|
||||
ix := indexOf(
|
||||
entryYAML("base-model", "acme/repo", "shared-weights-Q4_K_M.gguf", "aa"),
|
||||
entryYAML("base-model-heretic", "acme/repo", "shared-weights-Q8_0.gguf", "bb"),
|
||||
)
|
||||
ledger, err := ParseLedger([]byte("tokens:\n - {token: heretic, reason: \"finetune, not a re-quantization\"}\n"))
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
r := Propose(ix, ledger)
|
||||
Expect(suppressionReasons(r)).To(ContainSubstring("finetune, not a re-quantization"))
|
||||
Expect(suppressionReasons(r)).To(ContainSubstring(`token "heretic"`))
|
||||
})
|
||||
})
|
||||
|
||||
Describe("parent selection", func() {
|
||||
It("picks the bare-named entry when one exists", func() {
|
||||
ix := indexOf(
|
||||
entryYAML("base-model-q8_0", "acme/repo", "base-model-Q8_0.gguf", "aa"),
|
||||
entryYAML("base-model", "acme/repo", "base-model-Q4_K_M.gguf", "bb"),
|
||||
entryYAML("base-model-f16", "acme/repo", "base-model-f16.gguf", "cc"),
|
||||
)
|
||||
r := Propose(ix, nil)
|
||||
Expect(familyNames(r)).To(ConsistOf("base-model <- base-model-f16, base-model-q8_0"))
|
||||
})
|
||||
|
||||
It("picks the smallest build when no entry is bare-named", func() {
|
||||
ix := indexOf(
|
||||
entryYAML("ced-base-f16", "acme/repo", "ced-base-f16.gguf", "aa"),
|
||||
entryYAML("ced-base-q8", "acme/repo", "ced-base-Q8_0.gguf", "bb"),
|
||||
)
|
||||
r := Propose(ix, nil)
|
||||
Expect(familyNames(r)).To(ConsistOf("ced-base-q8 <- ced-base-f16"))
|
||||
})
|
||||
|
||||
It("judges the smallest build by the quantization in the model filename, not the name", func() {
|
||||
// The names carry no marker at all; only the filenames say which
|
||||
// build is which.
|
||||
ix := indexOf(
|
||||
entryYAML("thing-hq", "acme/repo", "thing-weights-BF16.gguf", "aa"),
|
||||
entryYAML("thing-lite", "acme/repo", "thing-weights-Q4_K_M.gguf", "bb"),
|
||||
)
|
||||
r := Propose(ix, nil)
|
||||
Expect(familyNames(r)).To(ConsistOf("thing-lite <- thing-hq"))
|
||||
})
|
||||
})
|
||||
|
||||
Describe("the rules a proposal has to respect", func() {
|
||||
It("refuses to nest: a target that already offers variants of its own", func() {
|
||||
ix := indexOf(
|
||||
entryYAML("nest-model", "acme/repo", "nest-model-Q4_K_M.gguf", "aa"),
|
||||
`- name: nest-model-q8_0
|
||||
url: u
|
||||
variants:
|
||||
- model: nest-model-q8_0-mtp
|
||||
overrides:
|
||||
parameters:
|
||||
model: nest-model-Q8_0.gguf
|
||||
files:
|
||||
- filename: nest-model-Q8_0.gguf
|
||||
uri: huggingface://acme/repo/nest-model-Q8_0.gguf
|
||||
sha256: bb
|
||||
`,
|
||||
entryYAML("nest-model-q8_0-mtp", "other/repo", "nest-model-mtp.gguf", "cc"),
|
||||
)
|
||||
r := Propose(ix, nil)
|
||||
Expect(familyNames(r)).To(BeEmpty())
|
||||
Expect(refusalReasons(r)).To(ContainSubstring("already offers variants of its own"))
|
||||
})
|
||||
|
||||
It("refuses to nest: a parent that is already somebody else's variant", func() {
|
||||
ix := indexOf(
|
||||
`- name: outer
|
||||
url: u
|
||||
variants:
|
||||
- model: middle
|
||||
overrides:
|
||||
parameters:
|
||||
model: outer-Q4_K_M.gguf
|
||||
files:
|
||||
- filename: outer-Q4_K_M.gguf
|
||||
uri: huggingface://acme/repo/outer-Q4_K_M.gguf
|
||||
sha256: aa
|
||||
`,
|
||||
entryYAML("middle", "acme/other", "middle-Q4_K_M.gguf", "bb"),
|
||||
entryYAML("middle-q8_0", "acme/other", "middle-Q8_0.gguf", "cc"),
|
||||
)
|
||||
r := Propose(ix, nil)
|
||||
Expect(familyNames(r)).To(BeEmpty())
|
||||
Expect(refusalReasons(r)).To(ContainSubstring("would nest variants"))
|
||||
})
|
||||
|
||||
It("refuses to let two parents claim one target", func() {
|
||||
ix := indexOf(
|
||||
`- name: claimant
|
||||
url: u
|
||||
variants:
|
||||
- model: contested-q8_0
|
||||
overrides:
|
||||
parameters:
|
||||
model: claimant-Q4_K_M.gguf
|
||||
files:
|
||||
- filename: claimant-Q4_K_M.gguf
|
||||
uri: huggingface://acme/repo/claimant-Q4_K_M.gguf
|
||||
sha256: aa
|
||||
`,
|
||||
entryYAML("contested", "acme/other", "contested-Q4_K_M.gguf", "bb"),
|
||||
entryYAML("contested-q8_0", "acme/other", "contested-Q8_0.gguf", "cc"),
|
||||
)
|
||||
r := Propose(ix, nil)
|
||||
Expect(familyNames(r)).To(BeEmpty())
|
||||
Expect(refusalReasons(r)).To(ContainSubstring("already a variant of"))
|
||||
})
|
||||
|
||||
It("refuses a target that is not independently installable", func() {
|
||||
ix := indexOf(
|
||||
entryYAML("stub-model", "acme/repo", "stub-model-Q4_K_M.gguf", "aa"),
|
||||
"- name: stub-model-q8_0\n description: a stanza nobody finished\n",
|
||||
)
|
||||
r := Propose(ix, nil)
|
||||
Expect(familyNames(r)).To(BeEmpty())
|
||||
Expect(refusalReasons(r)).To(ContainSubstring("not independently installable"))
|
||||
})
|
||||
|
||||
It("refuses a family whose parent defines a merge anchor, naming the entries that would inherit", func() {
|
||||
ix := indexOf(
|
||||
`- &anchored
|
||||
name: anchored-model
|
||||
url: u
|
||||
overrides:
|
||||
parameters:
|
||||
model: anchored-Q4_K_M.gguf
|
||||
files:
|
||||
- filename: anchored-Q4_K_M.gguf
|
||||
uri: huggingface://acme/repo/anchored-Q4_K_M.gguf
|
||||
sha256: aa
|
||||
`,
|
||||
`- !!merge <<: *anchored
|
||||
name: anchored-child
|
||||
variants: []
|
||||
overrides:
|
||||
parameters:
|
||||
model: unrelated-child-Q4_K_M.gguf
|
||||
files:
|
||||
- filename: unrelated-child-Q4_K_M.gguf
|
||||
uri: huggingface://other/repo/unrelated-child-Q4_K_M.gguf
|
||||
sha256: cc
|
||||
`,
|
||||
entryYAML("anchored-model-q8_0", "acme/repo", "anchored-Q8_0.gguf", "bb"),
|
||||
)
|
||||
r := Propose(ix, nil)
|
||||
Expect(familyNames(r)).To(BeEmpty())
|
||||
Expect(refusalReasons(r)).To(ContainSubstring("defines YAML anchor &anchored"))
|
||||
Expect(refusalReasons(r)).To(ContainSubstring("anchored-child"))
|
||||
Expect(refusalReasons(r)).To(ContainSubstring("variants: []"))
|
||||
})
|
||||
|
||||
It("refuses an entry whose name is not unique in the gallery", func() {
|
||||
ix := indexOf(
|
||||
entryYAML("twin", "acme/repo", "twin-Q4_K_M.gguf", "aa"),
|
||||
entryYAML("twin", "acme/repo", "twin-Q4_K_M.gguf", "aa"),
|
||||
entryYAML("twin-q8_0", "acme/repo", "twin-Q8_0.gguf", "bb"),
|
||||
)
|
||||
r := Propose(ix, nil)
|
||||
Expect(familyNames(r)).To(BeEmpty())
|
||||
Expect(refusalReasons(r)).To(ContainSubstring("appears more than once"))
|
||||
})
|
||||
|
||||
It("says nothing about a pair that is already grouped", func() {
|
||||
ix := indexOf(
|
||||
`- name: settled
|
||||
url: u
|
||||
variants:
|
||||
- model: settled-q8_0
|
||||
overrides:
|
||||
parameters:
|
||||
model: settled-Q4_K_M.gguf
|
||||
files:
|
||||
- filename: settled-Q4_K_M.gguf
|
||||
uri: huggingface://acme/repo/settled-Q4_K_M.gguf
|
||||
sha256: aa
|
||||
`,
|
||||
entryYAML("settled-q8_0", "acme/repo", "settled-Q8_0.gguf", "bb"),
|
||||
)
|
||||
r := Propose(ix, nil)
|
||||
Expect(r.HasProposals()).To(BeFalse())
|
||||
Expect(r.Refusals).To(BeEmpty())
|
||||
Expect(r.Suppressed).To(BeEmpty())
|
||||
})
|
||||
|
||||
It("adds only the missing members to a family that already exists", func() {
|
||||
ix := indexOf(
|
||||
`- name: partial
|
||||
url: u
|
||||
variants:
|
||||
- model: partial-q8_0
|
||||
overrides:
|
||||
parameters:
|
||||
model: partial-Q4_K_M.gguf
|
||||
files:
|
||||
- filename: partial-Q4_K_M.gguf
|
||||
uri: huggingface://acme/repo/partial-Q4_K_M.gguf
|
||||
sha256: aa
|
||||
`,
|
||||
entryYAML("partial-q8_0", "acme/repo", "partial-Q8_0.gguf", "bb"),
|
||||
entryYAML("partial-f16", "acme/repo", "partial-f16.gguf", "cc"),
|
||||
)
|
||||
r := Propose(ix, nil)
|
||||
Expect(familyNames(r)).To(ConsistOf("partial <- partial-f16"))
|
||||
})
|
||||
})
|
||||
|
||||
It("does not modify the index it was given", func() {
|
||||
text := entryYAML("foo-model", "acme/foo-GGUF", "foo-model-Q4_K_M.gguf", "aa") +
|
||||
entryYAML("foo-model-q8_0", "acme/foo-GGUF", "foo-model-Q8_0.gguf", "bb")
|
||||
ix, err := ParseIndex(text)
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
before := strings.Join(ix.Lines, "\n")
|
||||
Propose(ix, nil)
|
||||
Expect(strings.Join(ix.Lines, "\n")).To(Equal(before))
|
||||
})
|
||||
})
|
||||
151
.github/ci/variantproposals/quant.go
vendored
Normal file
151
.github/ci/variantproposals/quant.go
vendored
Normal file
@@ -0,0 +1,151 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"regexp"
|
||||
"strconv"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// Quantization and precision markers that distinguish one build of a set of
|
||||
// weights from another build of the same weights. Stripping them from a name
|
||||
// is what lets the proposer notice that two entries are the same model.
|
||||
//
|
||||
// qat and apex are in this list on a maintainer ruling: they are quantization
|
||||
// techniques applied to published weights, not separate weights. Names that use
|
||||
// "apex" to mean a finetune are handled by the rejection ledger instead, because
|
||||
// no amount of pattern matching can tell the two uses apart.
|
||||
const quantAlternation = `q[2-8](?:_[0-9a-z]+)*|pq[2-8](?:_[0-9a-z]+)*|iq[1-9][0-9a-z]*(?:_[0-9a-z]+)*|i1|` +
|
||||
`f16|f32|bf16|fp16|fp32|fp8|fp4|nvfp4|mxfp4(?:_moe)*|awq|gptq|qat|apex|gguf|ggml|[0-9]+bit|g[0-9]+`
|
||||
|
||||
// quantSegment matches a whole hyphen-delimited segment of an entry name.
|
||||
// Names separate their parts with "-" and keep quantization tokens internally
|
||||
// joined with "_", so a segment is the right unit here: "q4_k_m" arrives whole.
|
||||
var quantSegment = regexp.MustCompile(`^(?:` + quantAlternation + `)$`)
|
||||
|
||||
// quantFileSuffix matches a trailing quantization token in a weight filename.
|
||||
// Filenames mix "-", "_" and "." as separators, so unlike entry names they
|
||||
// cannot be split into segments up front without tearing "Q4_K_M" apart.
|
||||
var quantFileSuffix = regexp.MustCompile(`(?i)[-_.](?:` + quantAlternation + `)$`)
|
||||
|
||||
var weightExtension = regexp.MustCompile(`(?i)\.(gguf|ggml|safetensors|bin|pt|pth|onnx)$`)
|
||||
|
||||
// IsQuantToken reports whether a single name segment is a quantization or
|
||||
// precision marker rather than part of the model's identity.
|
||||
func IsQuantToken(segment string) bool {
|
||||
return quantSegment.MatchString(strings.ToLower(segment))
|
||||
}
|
||||
|
||||
// NameStem reduces an entry name to the identity it shares with its alternative
|
||||
// builds: the config suffix after ":" is dropped, then trailing quantization
|
||||
// segments are stripped.
|
||||
//
|
||||
// It implements the first two grouping signals together because they answer the
|
||||
// same question. "foo:q8_0" and "foo-q8_0" are both alternative builds of "foo",
|
||||
// and the caller that needs to report which convention was used can compare the
|
||||
// name against the stem itself.
|
||||
//
|
||||
// At least one segment always survives, so a name made entirely of quantization
|
||||
// tokens does not collapse to the empty stem and swallow every other such name.
|
||||
func NameStem(name string) string {
|
||||
base := strings.ToLower(strings.TrimSpace(name))
|
||||
if i := strings.Index(base, ":"); i >= 0 {
|
||||
base = base[:i]
|
||||
}
|
||||
segments := strings.Split(base, "-")
|
||||
for len(segments) > 1 && quantSegment.MatchString(segments[len(segments)-1]) {
|
||||
segments = segments[:len(segments)-1]
|
||||
}
|
||||
return strings.Join(segments, "-")
|
||||
}
|
||||
|
||||
// HasConfigSuffix reports whether a name uses the ":" convention for naming a
|
||||
// config variant of another entry.
|
||||
func HasConfigSuffix(name string) bool {
|
||||
return strings.Contains(name, ":")
|
||||
}
|
||||
|
||||
// FileStem reduces a weight filename to the identity shared by its other
|
||||
// quantizations: directories, extension and trailing quantization tokens go.
|
||||
//
|
||||
// This is the third grouping signal. It is the one that has misfired before, so
|
||||
// callers must filter auxiliary files out before handing a filename here: a
|
||||
// shared text encoder is not evidence of shared weights.
|
||||
func FileStem(filename string) string {
|
||||
base := filename
|
||||
if i := strings.LastIndex(base, "/"); i >= 0 {
|
||||
base = base[i+1:]
|
||||
}
|
||||
base = weightExtension.ReplaceAllString(base, "")
|
||||
for {
|
||||
stripped := quantFileSuffix.ReplaceAllString(base, "")
|
||||
if stripped == base {
|
||||
break
|
||||
}
|
||||
base = stripped
|
||||
}
|
||||
return strings.ToLower(base)
|
||||
}
|
||||
|
||||
// bitsPerWeight ranks quantization tokens so the smallest build of a family can
|
||||
// be identified when no bare-named entry exists to be the parent.
|
||||
//
|
||||
// The figures are nominal bits per weight, not measured file sizes. Ranking is
|
||||
// all that is asked of them, and a nominal figure is available from the name
|
||||
// alone without downloading anything.
|
||||
func bitsPerWeight(token string) (int, bool) {
|
||||
t := strings.ToLower(token)
|
||||
switch {
|
||||
case t == "i1":
|
||||
return 1, true
|
||||
case strings.HasPrefix(t, "nvfp4"), strings.HasPrefix(t, "mxfp4"), t == "fp4":
|
||||
return 4, true
|
||||
case t == "fp8":
|
||||
return 8, true
|
||||
case t == "f16", t == "bf16", t == "fp16":
|
||||
return 16, true
|
||||
case t == "f32", t == "fp32":
|
||||
return 32, true
|
||||
case t == "awq", t == "gptq":
|
||||
return 4, true
|
||||
}
|
||||
if m := regexp.MustCompile(`^p?q([1-9])`).FindStringSubmatch(t); m != nil {
|
||||
n, _ := strconv.Atoi(m[1])
|
||||
return n, true
|
||||
}
|
||||
if m := regexp.MustCompile(`^iq([1-9])`).FindStringSubmatch(t); m != nil {
|
||||
n, _ := strconv.Atoi(m[1])
|
||||
return n, true
|
||||
}
|
||||
if m := regexp.MustCompile(`^([0-9]+)bit$`).FindStringSubmatch(t); m != nil {
|
||||
n, _ := strconv.Atoi(m[1])
|
||||
return n, true
|
||||
}
|
||||
return 0, false
|
||||
}
|
||||
|
||||
// unknownWidth sorts after every recognised quantization so an entry whose
|
||||
// build cannot be read from its filename never wins the "smallest build" tie
|
||||
// break by accident.
|
||||
const unknownWidth = 1 << 10
|
||||
|
||||
// BuildWidth reports the nominal bits per weight of the build a filename holds.
|
||||
// An unreadable filename gets unknownWidth.
|
||||
func BuildWidth(filename string) int {
|
||||
base := filename
|
||||
if i := strings.LastIndex(base, "/"); i >= 0 {
|
||||
base = base[i+1:]
|
||||
}
|
||||
base = weightExtension.ReplaceAllString(base, "")
|
||||
best := unknownWidth
|
||||
for {
|
||||
m := quantFileSuffix.FindString(base)
|
||||
if m == "" {
|
||||
break
|
||||
}
|
||||
if bits, ok := bitsPerWeight(m[1:]); ok && bits < best {
|
||||
best = bits
|
||||
}
|
||||
base = base[:len(base)-len(m)]
|
||||
}
|
||||
return best
|
||||
}
|
||||
92
.github/ci/variantproposals/quant_test.go
vendored
Normal file
92
.github/ci/variantproposals/quant_test.go
vendored
Normal file
@@ -0,0 +1,92 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
)
|
||||
|
||||
var _ = Describe("quantization markers", func() {
|
||||
DescribeTable("NameStem strips the markers that distinguish builds, not models",
|
||||
func(name, expected string) {
|
||||
Expect(NameStem(name)).To(Equal(expected))
|
||||
},
|
||||
Entry("plain q4", "foo-model-q4_k_m", "foo-model"),
|
||||
Entry("q8_0", "foo-model-q8_0", "foo-model"),
|
||||
Entry("q5_1", "foo-model-q5_1", "foo-model"),
|
||||
Entry("q2 with group size", "ternary-bonsai-8b-q2-g64", "ternary-bonsai-8b"),
|
||||
Entry("iq variant", "ideogram-4-iq4nl-ggml", "ideogram-4"),
|
||||
Entry("i1 imatrix", "orca-agent-v0.1-i1", "orca-agent-v0.1"),
|
||||
Entry("f16", "ced-base-f16", "ced-base"),
|
||||
Entry("bf16", "some-model-bf16", "some-model"),
|
||||
Entry("fp8", "some-model-fp8", "some-model"),
|
||||
Entry("nvfp4", "qwen3.6-27b-nvfp4", "qwen3.6-27b"),
|
||||
Entry("mxfp4_moe", "huihui-qwen3-vl-30b-a3b-instruct-abliterated-mxfp4_moe", "huihui-qwen3-vl-30b-a3b-instruct-abliterated"),
|
||||
Entry("pq2", "ternary-bonsai-8b-pq2", "ternary-bonsai-8b"),
|
||||
Entry("awq", "some-model-awq", "some-model"),
|
||||
Entry("gptq", "some-model-gptq", "some-model"),
|
||||
Entry("Nbit", "qwen3-8b-mlx-4bit", "qwen3-8b-mlx"),
|
||||
Entry("gguf", "some-model-gguf", "some-model"),
|
||||
Entry("ggml", "flux.1-dev-ggml", "flux.1-dev"),
|
||||
Entry("qat is a quantization technique", "gemma-3-27b-it-qat", "gemma-3-27b-it"),
|
||||
Entry("apex is a quantization technique", "qwen3.6-35b-a3b-apex", "qwen3.6-35b-a3b"),
|
||||
Entry("stacked markers", "gemma-4-e2b-it-qat-q4_0", "gemma-4-e2b-it"),
|
||||
Entry("the config suffix is dropped", "phi-2-chat:Q8_0", "phi-2-chat"),
|
||||
Entry("a non-quant config suffix is dropped too", "meta-llama-3.1-8b-instruct:grammar-functioncall", "meta-llama-3.1-8b-instruct"),
|
||||
)
|
||||
|
||||
DescribeTable("NameStem leaves alone what identifies a different model",
|
||||
func(name, expected string) {
|
||||
Expect(NameStem(name)).To(Equal(expected))
|
||||
},
|
||||
Entry("parameter size", "qwen3-tts-cpp-0.6b-base", "qwen3-tts-cpp-0.6b-base"),
|
||||
Entry("language suffix", "kokoros-de", "kokoros-de"),
|
||||
Entry("English-only ASR", "whisper-small-en", "whisper-small-en"),
|
||||
Entry("finetune", "qwen3-30b-a3b-abliterated", "qwen3-30b-a3b-abliterated"),
|
||||
Entry("product suffix", "vibevoice-cpp-asr", "vibevoice-cpp-asr"),
|
||||
)
|
||||
|
||||
It("never strips a name down to nothing", func() {
|
||||
Expect(NameStem("q4_k_m")).To(Equal("q4_k_m"))
|
||||
Expect(NameStem("f16-q8_0")).To(Equal("f16"))
|
||||
})
|
||||
|
||||
DescribeTable("FileStem reduces a weight filename to the weights it holds",
|
||||
func(filename, expected string) {
|
||||
Expect(FileStem(filename)).To(Equal(expected))
|
||||
},
|
||||
Entry("directory and extension go", "bonsai/models/Ternary-Bonsai-8B-gguf/Ternary-Bonsai-8B-Q2_0.gguf", "ternary-bonsai-8b"),
|
||||
Entry("underscored quant token stays whole", "Llama-3.2-1B-Instruct-Q4_K_M.gguf", "llama-3.2-1b-instruct"),
|
||||
Entry("dot separated quant token", "Llama-3.2-3B-Instruct.Q4_K_M.gguf", "llama-3.2-3b-instruct"),
|
||||
Entry("group size suffix", "Ternary-Bonsai-8B-Q2_0_g64.gguf", "ternary-bonsai-8b"),
|
||||
Entry("bf16", "omnivoice-cpp-hq/omnivoice-base-BF16.gguf", "omnivoice-base"),
|
||||
Entry("safetensors", "some/dir/Model-Name-fp8.safetensors", "model-name"),
|
||||
)
|
||||
|
||||
DescribeTable("BuildWidth reads the nominal width out of a filename",
|
||||
func(filename string, expected int) {
|
||||
Expect(BuildWidth(filename)).To(Equal(expected))
|
||||
},
|
||||
Entry("q4", "foo-Q4_K_M.gguf", 4),
|
||||
Entry("q8", "foo-Q8_0.gguf", 8),
|
||||
Entry("q2", "foo-Q2_0.gguf", 2),
|
||||
Entry("f16", "foo-f16.gguf", 16),
|
||||
Entry("bf16", "foo-BF16.gguf", 16),
|
||||
Entry("iq3", "foo-iq3_xxs.gguf", 3),
|
||||
Entry("nothing readable sorts last", "foo.gguf", unknownWidth),
|
||||
)
|
||||
|
||||
It("treats an auxiliary file as never being the model's own weights", func() {
|
||||
for _, f := range []string{
|
||||
"mmproj-model-f16.gguf",
|
||||
"dir/vae-BF16.gguf",
|
||||
"clip_l.safetensors",
|
||||
"umt5-xxl-encoder-Q8_0.gguf",
|
||||
"t5xxl_fp16.safetensors",
|
||||
"ae.safetensors",
|
||||
"omnivoice-tokenizer-Q8_0.gguf",
|
||||
} {
|
||||
Expect(IsAuxiliaryFile(f)).To(BeTrue(), "expected %q to be auxiliary", f)
|
||||
}
|
||||
Expect(IsAuxiliaryFile("gemma-3-27b-it-Q4_K_M.gguf")).To(BeFalse())
|
||||
})
|
||||
})
|
||||
13
.github/ci/variantproposals/variantproposals_suite_test.go
vendored
Normal file
13
.github/ci/variantproposals/variantproposals_suite_test.go
vendored
Normal file
@@ -0,0 +1,13 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
)
|
||||
|
||||
func TestVariantProposals(t *testing.T) {
|
||||
RegisterFailHandler(Fail)
|
||||
RunSpecs(t, "gallery variant proposals")
|
||||
}
|
||||
10
.github/dependabot.yml
vendored
10
.github/dependabot.yml
vendored
@@ -45,6 +45,16 @@ updates:
|
||||
directory: "/backend/python/diffusers"
|
||||
schedule:
|
||||
interval: "weekly"
|
||||
# torch and transformers are deliberately pinned in this backend (see
|
||||
# backend/python/diffusers/requirements-*.txt and issue #9979), and the
|
||||
# l4t12 variant resolves them from the Jetson pip index
|
||||
# (https://pypi.jetson-ai-lab.io/jp6/cu129/). dependabot cannot authenticate
|
||||
# against that index and fails the whole weekly update with a
|
||||
# private_source_authentication_failure. Ignore the two pinned deps we don't
|
||||
# want bumped anyway so the job stays green.
|
||||
ignore:
|
||||
- dependency-name: "torch"
|
||||
- dependency-name: "transformers"
|
||||
- package-ecosystem: "pip"
|
||||
directory: "/backend/python/exllama"
|
||||
schedule:
|
||||
|
||||
77
.github/scripts/paged-canary-apply.sh
vendored
77
.github/scripts/paged-canary-apply.sh
vendored
@@ -1,77 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# paged-canary-apply.sh - apply the vendored paged-attention patch series
|
||||
# (backend/cpp/llama-cpp-localai-paged/patches/paged/0001-0030) to a llama.cpp checkout, the
|
||||
# same way the build does, but tolerating the ONE known-benign pre-existing
|
||||
# quirk in the series. Used by the early-warning canary
|
||||
# (.github/workflows/llama-cpp-paged-canary.yml) so it only goes red on a REAL
|
||||
# upstream break, never on that quirk.
|
||||
#
|
||||
# Usage: paged-canary-apply.sh <llama.cpp-checkout-dir> <patches-dir>
|
||||
# <patches-dir> is normally backend/cpp/llama-cpp-localai-paged/patches (it holds the
|
||||
# top-level base series 0*.patch, currently empty, and the paged/ subseries).
|
||||
#
|
||||
# Exit 0 = the whole series applied -> patches still fit upstream.
|
||||
# Exit !=0 = a patch failed to apply = the red signal: an upstream change moved
|
||||
# the tree out from under the patches, so it is time to run a PIN_SYNC.
|
||||
#
|
||||
# Apply method MIRRORS backend/cpp/llama-cpp/Makefile's `llama.cpp` target:
|
||||
# plain `git apply --verbose`, which natively tolerates @@ line-number offsets
|
||||
# but NOT context-line changes. Matching the build's method is the point - the
|
||||
# canary's apply result is exactly what the real build's apply would do.
|
||||
#
|
||||
# The ONLY tolerance, and it is path-scoped (not a blanket `|| true`): patch
|
||||
# 0019 carries a stray *modify* hunk against the dev-only doc
|
||||
# SSM_DECODE_FIX_RESULTS.md, a file that exists only on the DGX dev tree and is
|
||||
# absent from any clean upstream checkout. `git apply` is atomic, so that single
|
||||
# missing-file hunk rejects the whole patch - and because 0021/0022/0026/0028
|
||||
# build on 0019's code, the rejection cascades to them too. This is a
|
||||
# PRE-EXISTING shipped-series defect, present identically on every pin, NOT an
|
||||
# upstream break (see backend/cpp/llama-cpp-localai-paged/README.md section 7,
|
||||
# "Pin + maintenance policy"). We exclude ONLY that dev-doc path and still
|
||||
# apply 0019's real code hunks atomically, so a genuine code-hunk break in 0019
|
||||
# still fails the canary. prepare.sh tolerates the same hunk via
|
||||
# `patch ... || true`; this mirrors that tolerance precisely.
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
CHECKOUT="${1:?usage: paged-canary-apply.sh <llama.cpp-checkout> <patches-dir>}"
|
||||
PATCHES="${2:?usage: paged-canary-apply.sh <llama.cpp-checkout> <patches-dir>}"
|
||||
|
||||
# The lone tolerated dev-doc, and the only patch allowed to carry it.
|
||||
DEVDOC_GLOB='*SSM_DECODE_FIX_RESULTS.md'
|
||||
DEVDOC_PATCH='0019-qwen35-ssm-decode-fused-gather.patch'
|
||||
|
||||
# Resolve to absolute paths so the apply works after we cd into the checkout.
|
||||
PATCHES="$(cd "$PATCHES" && pwd)"
|
||||
cd "$CHECKOUT"
|
||||
|
||||
shopt -s nullglob
|
||||
|
||||
apply_one() {
|
||||
local p="$1"; shift
|
||||
echo "paged-canary: applying $(basename "$p")"
|
||||
if ! git apply --verbose "$@" "$p"; then
|
||||
echo "::error::paged patch no longer applies to the upstream llama.cpp tip: $(basename "$p")"
|
||||
echo "::error::upstream drifted past the vendored paged series - run a PIN_SYNC (see backend/cpp/llama-cpp-localai-paged/README.md section 7, Pin + maintenance policy), do NOT bump the pin blindly"
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
|
||||
# Base series first (parity with the build: patches/0*.patch before
|
||||
# patches/paged/0*.patch). Currently empty; nullglob makes this a no-op.
|
||||
for p in "$PATCHES"/0*.patch; do
|
||||
apply_one "$p"
|
||||
done
|
||||
|
||||
# Paged series, in order.
|
||||
for p in "$PATCHES"/paged/0*.patch; do
|
||||
if [ "$(basename "$p")" = "$DEVDOC_PATCH" ]; then
|
||||
# Apply 0019's real code hunks; exclude ONLY the benign dev-doc hunk.
|
||||
apply_one "$p" --exclude="$DEVDOC_GLOB"
|
||||
else
|
||||
apply_one "$p"
|
||||
fi
|
||||
done
|
||||
|
||||
echo "paged-canary: the full paged patch series applied cleanly to the upstream tip"
|
||||
181
.github/workflows/backend.yml
vendored
181
.github/workflows/backend.yml
vendored
@@ -32,16 +32,30 @@ jobs:
|
||||
if: github.repository == 'mudler/LocalAI'
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
matrix-singlearch: ${{ steps.set-matrix.outputs['matrix-singlearch'] }}
|
||||
matrix-multiarch: ${{ steps.set-matrix.outputs['matrix-multiarch'] }}
|
||||
matrix-darwin: ${{ steps.set-matrix.outputs['matrix-darwin'] }}
|
||||
merge-matrix-multiarch: ${{ steps.set-matrix.outputs['merge-matrix-multiarch'] }}
|
||||
merge-matrix-singlearch: ${{ steps.set-matrix.outputs['merge-matrix-singlearch'] }}
|
||||
has-backends-singlearch: ${{ steps.set-matrix.outputs['has-backends-singlearch'] }}
|
||||
has-backends-multiarch: ${{ steps.set-matrix.outputs['has-backends-multiarch'] }}
|
||||
has-backends-darwin: ${{ steps.set-matrix.outputs['has-backends-darwin'] }}
|
||||
has-merges-multiarch: ${{ steps.set-matrix.outputs['has-merges-multiarch'] }}
|
||||
has-merges-singlearch: ${{ steps.set-matrix.outputs['has-merges-singlearch'] }}
|
||||
# Single-arch backends are sharded across SINGLEARCH_SHARDS matrix jobs to
|
||||
# stay under GitHub's 256-jobs-per-matrix limit (see changed-backends.js).
|
||||
matrix-singlearch-1: ${{ steps.set-matrix.outputs['matrix-singlearch-1'] }}
|
||||
merge-matrix-singlearch-1: ${{ steps.set-matrix.outputs['merge-matrix-singlearch-1'] }}
|
||||
has-backends-singlearch-1: ${{ steps.set-matrix.outputs['has-backends-singlearch-1'] }}
|
||||
has-merges-singlearch-1: ${{ steps.set-matrix.outputs['has-merges-singlearch-1'] }}
|
||||
matrix-singlearch-2: ${{ steps.set-matrix.outputs['matrix-singlearch-2'] }}
|
||||
merge-matrix-singlearch-2: ${{ steps.set-matrix.outputs['merge-matrix-singlearch-2'] }}
|
||||
has-backends-singlearch-2: ${{ steps.set-matrix.outputs['has-backends-singlearch-2'] }}
|
||||
has-merges-singlearch-2: ${{ steps.set-matrix.outputs['has-merges-singlearch-2'] }}
|
||||
matrix-singlearch-3: ${{ steps.set-matrix.outputs['matrix-singlearch-3'] }}
|
||||
merge-matrix-singlearch-3: ${{ steps.set-matrix.outputs['merge-matrix-singlearch-3'] }}
|
||||
has-backends-singlearch-3: ${{ steps.set-matrix.outputs['has-backends-singlearch-3'] }}
|
||||
has-merges-singlearch-3: ${{ steps.set-matrix.outputs['has-merges-singlearch-3'] }}
|
||||
matrix-singlearch-4: ${{ steps.set-matrix.outputs['matrix-singlearch-4'] }}
|
||||
merge-matrix-singlearch-4: ${{ steps.set-matrix.outputs['merge-matrix-singlearch-4'] }}
|
||||
has-backends-singlearch-4: ${{ steps.set-matrix.outputs['has-backends-singlearch-4'] }}
|
||||
has-merges-singlearch-4: ${{ steps.set-matrix.outputs['has-merges-singlearch-4'] }}
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v7
|
||||
@@ -109,9 +123,9 @@ jobs:
|
||||
# take their full ~6h cold without blocking manifest assembly for the
|
||||
# multi-arch backends whose per-arch digests would otherwise sit untagged
|
||||
# on quay long enough to be GC'd.
|
||||
backend-jobs-singlearch:
|
||||
backend-jobs-singlearch-1:
|
||||
needs: generate-matrix
|
||||
if: needs.generate-matrix.outputs['has-backends-singlearch'] == 'true'
|
||||
if: needs.generate-matrix.outputs['has-backends-singlearch-1'] == 'true'
|
||||
uses: ./.github/workflows/backend_build.yml
|
||||
with:
|
||||
tag-latest: ${{ matrix.tag-latest }}
|
||||
@@ -138,7 +152,100 @@ jobs:
|
||||
strategy:
|
||||
fail-fast: false
|
||||
max-parallel: 8
|
||||
matrix: ${{ fromJson(needs.generate-matrix.outputs['matrix-singlearch']) }}
|
||||
matrix: ${{ fromJson(needs.generate-matrix.outputs['matrix-singlearch-1']) }}
|
||||
|
||||
backend-jobs-singlearch-2:
|
||||
needs: generate-matrix
|
||||
if: needs.generate-matrix.outputs['has-backends-singlearch-2'] == 'true'
|
||||
uses: ./.github/workflows/backend_build.yml
|
||||
with:
|
||||
tag-latest: ${{ matrix.tag-latest }}
|
||||
tag-suffix: ${{ matrix.tag-suffix }}
|
||||
build-type: ${{ matrix.build-type }}
|
||||
cuda-major-version: ${{ matrix.cuda-major-version }}
|
||||
cuda-minor-version: ${{ matrix.cuda-minor-version }}
|
||||
platforms: ${{ matrix.platforms }}
|
||||
platform-tag: ${{ matrix.platform-tag || '' }}
|
||||
runs-on: ${{ matrix.runs-on }}
|
||||
builder-base-image: ${{ matrix.builder-base-image || '' }}
|
||||
base-image: ${{ matrix.base-image }}
|
||||
backend: ${{ matrix.backend }}
|
||||
dockerfile: ${{ matrix.dockerfile }}
|
||||
skip-drivers: ${{ matrix.skip-drivers }}
|
||||
context: ${{ matrix.context }}
|
||||
ubuntu-version: ${{ matrix.ubuntu-version }}
|
||||
amdgpu-targets: ${{ matrix.amdgpu-targets || 'gfx908,gfx90a,gfx942,gfx950,gfx1030,gfx1100,gfx1101,gfx1102,gfx1151,gfx1200,gfx1201' }}
|
||||
secrets:
|
||||
dockerUsername: ${{ secrets.DOCKERHUB_USERNAME }}
|
||||
dockerPassword: ${{ secrets.DOCKERHUB_PASSWORD }}
|
||||
quayUsername: ${{ secrets.LOCALAI_REGISTRY_USERNAME }}
|
||||
quayPassword: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
max-parallel: 8
|
||||
matrix: ${{ fromJson(needs.generate-matrix.outputs['matrix-singlearch-2']) }}
|
||||
|
||||
backend-jobs-singlearch-3:
|
||||
needs: generate-matrix
|
||||
if: needs.generate-matrix.outputs['has-backends-singlearch-3'] == 'true'
|
||||
uses: ./.github/workflows/backend_build.yml
|
||||
with:
|
||||
tag-latest: ${{ matrix.tag-latest }}
|
||||
tag-suffix: ${{ matrix.tag-suffix }}
|
||||
build-type: ${{ matrix.build-type }}
|
||||
cuda-major-version: ${{ matrix.cuda-major-version }}
|
||||
cuda-minor-version: ${{ matrix.cuda-minor-version }}
|
||||
platforms: ${{ matrix.platforms }}
|
||||
platform-tag: ${{ matrix.platform-tag || '' }}
|
||||
runs-on: ${{ matrix.runs-on }}
|
||||
builder-base-image: ${{ matrix.builder-base-image || '' }}
|
||||
base-image: ${{ matrix.base-image }}
|
||||
backend: ${{ matrix.backend }}
|
||||
dockerfile: ${{ matrix.dockerfile }}
|
||||
skip-drivers: ${{ matrix.skip-drivers }}
|
||||
context: ${{ matrix.context }}
|
||||
ubuntu-version: ${{ matrix.ubuntu-version }}
|
||||
amdgpu-targets: ${{ matrix.amdgpu-targets || 'gfx908,gfx90a,gfx942,gfx950,gfx1030,gfx1100,gfx1101,gfx1102,gfx1151,gfx1200,gfx1201' }}
|
||||
secrets:
|
||||
dockerUsername: ${{ secrets.DOCKERHUB_USERNAME }}
|
||||
dockerPassword: ${{ secrets.DOCKERHUB_PASSWORD }}
|
||||
quayUsername: ${{ secrets.LOCALAI_REGISTRY_USERNAME }}
|
||||
quayPassword: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
max-parallel: 8
|
||||
matrix: ${{ fromJson(needs.generate-matrix.outputs['matrix-singlearch-3']) }}
|
||||
|
||||
backend-jobs-singlearch-4:
|
||||
needs: generate-matrix
|
||||
if: needs.generate-matrix.outputs['has-backends-singlearch-4'] == 'true'
|
||||
uses: ./.github/workflows/backend_build.yml
|
||||
with:
|
||||
tag-latest: ${{ matrix.tag-latest }}
|
||||
tag-suffix: ${{ matrix.tag-suffix }}
|
||||
build-type: ${{ matrix.build-type }}
|
||||
cuda-major-version: ${{ matrix.cuda-major-version }}
|
||||
cuda-minor-version: ${{ matrix.cuda-minor-version }}
|
||||
platforms: ${{ matrix.platforms }}
|
||||
platform-tag: ${{ matrix.platform-tag || '' }}
|
||||
runs-on: ${{ matrix.runs-on }}
|
||||
builder-base-image: ${{ matrix.builder-base-image || '' }}
|
||||
base-image: ${{ matrix.base-image }}
|
||||
backend: ${{ matrix.backend }}
|
||||
dockerfile: ${{ matrix.dockerfile }}
|
||||
skip-drivers: ${{ matrix.skip-drivers }}
|
||||
context: ${{ matrix.context }}
|
||||
ubuntu-version: ${{ matrix.ubuntu-version }}
|
||||
amdgpu-targets: ${{ matrix.amdgpu-targets || 'gfx908,gfx90a,gfx942,gfx950,gfx1030,gfx1100,gfx1101,gfx1102,gfx1151,gfx1200,gfx1201' }}
|
||||
secrets:
|
||||
dockerUsername: ${{ secrets.DOCKERHUB_USERNAME }}
|
||||
dockerPassword: ${{ secrets.DOCKERHUB_PASSWORD }}
|
||||
quayUsername: ${{ secrets.LOCALAI_REGISTRY_USERNAME }}
|
||||
quayPassword: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
max-parallel: 8
|
||||
matrix: ${{ fromJson(needs.generate-matrix.outputs['matrix-singlearch-4']) }}
|
||||
|
||||
# Apply tags to per-arch digests via `imagetools create`. Split into two
|
||||
# jobs that mirror the build split so each merge waits ONLY on its
|
||||
@@ -174,10 +281,12 @@ jobs:
|
||||
fail-fast: false
|
||||
matrix: ${{ fromJson(needs.generate-matrix.outputs['merge-matrix-multiarch']) }}
|
||||
|
||||
backend-merge-jobs-singlearch:
|
||||
needs: [generate-matrix, backend-jobs-singlearch]
|
||||
# See note on backend-merge-jobs-multiarch above for !cancelled().
|
||||
if: ${{ !cancelled() && needs.generate-matrix.outputs['has-merges-singlearch'] == 'true' }}
|
||||
# One merge shard per build shard: backend-merge-jobs-singlearch-<n> needs only
|
||||
# backend-jobs-singlearch-<n>, preserving the "merge waits only on its own
|
||||
# build" property while staying under the 256-jobs-per-matrix limit.
|
||||
backend-merge-jobs-singlearch-1:
|
||||
needs: [generate-matrix, backend-jobs-singlearch-1]
|
||||
if: ${{ !cancelled() && needs.generate-matrix.outputs['has-merges-singlearch-1'] == 'true' }}
|
||||
uses: ./.github/workflows/backend_merge.yml
|
||||
with:
|
||||
tag-latest: ${{ matrix.tag-latest }}
|
||||
@@ -189,7 +298,55 @@ jobs:
|
||||
quayPassword: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix: ${{ fromJson(needs.generate-matrix.outputs['merge-matrix-singlearch']) }}
|
||||
matrix: ${{ fromJson(needs.generate-matrix.outputs['merge-matrix-singlearch-1']) }}
|
||||
|
||||
backend-merge-jobs-singlearch-2:
|
||||
needs: [generate-matrix, backend-jobs-singlearch-2]
|
||||
if: ${{ !cancelled() && needs.generate-matrix.outputs['has-merges-singlearch-2'] == 'true' }}
|
||||
uses: ./.github/workflows/backend_merge.yml
|
||||
with:
|
||||
tag-latest: ${{ matrix.tag-latest }}
|
||||
tag-suffix: ${{ matrix.tag-suffix }}
|
||||
secrets:
|
||||
dockerUsername: ${{ secrets.DOCKERHUB_USERNAME }}
|
||||
dockerPassword: ${{ secrets.DOCKERHUB_PASSWORD }}
|
||||
quayUsername: ${{ secrets.LOCALAI_REGISTRY_USERNAME }}
|
||||
quayPassword: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix: ${{ fromJson(needs.generate-matrix.outputs['merge-matrix-singlearch-2']) }}
|
||||
|
||||
backend-merge-jobs-singlearch-3:
|
||||
needs: [generate-matrix, backend-jobs-singlearch-3]
|
||||
if: ${{ !cancelled() && needs.generate-matrix.outputs['has-merges-singlearch-3'] == 'true' }}
|
||||
uses: ./.github/workflows/backend_merge.yml
|
||||
with:
|
||||
tag-latest: ${{ matrix.tag-latest }}
|
||||
tag-suffix: ${{ matrix.tag-suffix }}
|
||||
secrets:
|
||||
dockerUsername: ${{ secrets.DOCKERHUB_USERNAME }}
|
||||
dockerPassword: ${{ secrets.DOCKERHUB_PASSWORD }}
|
||||
quayUsername: ${{ secrets.LOCALAI_REGISTRY_USERNAME }}
|
||||
quayPassword: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix: ${{ fromJson(needs.generate-matrix.outputs['merge-matrix-singlearch-3']) }}
|
||||
|
||||
backend-merge-jobs-singlearch-4:
|
||||
needs: [generate-matrix, backend-jobs-singlearch-4]
|
||||
if: ${{ !cancelled() && needs.generate-matrix.outputs['has-merges-singlearch-4'] == 'true' }}
|
||||
uses: ./.github/workflows/backend_merge.yml
|
||||
with:
|
||||
tag-latest: ${{ matrix.tag-latest }}
|
||||
tag-suffix: ${{ matrix.tag-suffix }}
|
||||
secrets:
|
||||
dockerUsername: ${{ secrets.DOCKERHUB_USERNAME }}
|
||||
dockerPassword: ${{ secrets.DOCKERHUB_PASSWORD }}
|
||||
quayUsername: ${{ secrets.LOCALAI_REGISTRY_USERNAME }}
|
||||
quayPassword: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix: ${{ fromJson(needs.generate-matrix.outputs['merge-matrix-singlearch-4']) }}
|
||||
|
||||
backend-jobs-darwin:
|
||||
needs: generate-matrix
|
||||
|
||||
25
.github/workflows/backend_build_darwin.yml
vendored
25
.github/workflows/backend_build_darwin.yml
vendored
@@ -82,7 +82,7 @@ jobs:
|
||||
# as the Linux registry cache.
|
||||
- name: Restore Homebrew cache
|
||||
id: brew-cache
|
||||
uses: actions/cache/restore@v4
|
||||
uses: actions/cache/restore@v6
|
||||
with:
|
||||
path: |
|
||||
~/Library/Caches/Homebrew/downloads
|
||||
@@ -142,7 +142,7 @@ jobs:
|
||||
|
||||
- name: Save Homebrew cache
|
||||
if: github.event_name != 'pull_request' && steps.brew-cache.outputs.cache-hit != 'true'
|
||||
uses: actions/cache/save@v4
|
||||
uses: actions/cache/save@v6
|
||||
with:
|
||||
path: |
|
||||
~/Library/Caches/Homebrew/downloads
|
||||
@@ -169,16 +169,16 @@ jobs:
|
||||
# invalidates cleanly; restore-keys fall back to the latest entry for the
|
||||
# same pin so unchanged TUs stay warm even when the cache is fresh.
|
||||
- name: Compute llama.cpp version
|
||||
if: inputs.backend == 'llama-cpp' || inputs.backend == 'llama-cpp-localai-paged'
|
||||
if: inputs.backend == 'llama-cpp'
|
||||
id: llama-version
|
||||
run: |
|
||||
version=$(grep '^LLAMA_VERSION' backend/cpp/llama-cpp/Makefile | head -1 | cut -d= -f2 | cut -d'?' -f1 | tr -d ' ')
|
||||
echo "version=${version}" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Restore ccache
|
||||
if: inputs.backend == 'llama-cpp' || inputs.backend == 'llama-cpp-localai-paged'
|
||||
if: inputs.backend == 'llama-cpp'
|
||||
id: ccache-cache
|
||||
uses: actions/cache/restore@v4
|
||||
uses: actions/cache/restore@v6
|
||||
with:
|
||||
path: ~/Library/Caches/ccache
|
||||
key: ccache-llama-${{ runner.arch }}-${{ steps.llama-version.outputs.version }}-${{ github.run_id }}
|
||||
@@ -186,7 +186,7 @@ jobs:
|
||||
ccache-llama-${{ runner.arch }}-${{ steps.llama-version.outputs.version }}-
|
||||
|
||||
- name: Configure ccache
|
||||
if: inputs.backend == 'llama-cpp' || inputs.backend == 'llama-cpp-localai-paged'
|
||||
if: inputs.backend == 'llama-cpp'
|
||||
run: |
|
||||
mkdir -p "$HOME/Library/Caches/ccache"
|
||||
ccache -M 2G
|
||||
@@ -211,7 +211,7 @@ jobs:
|
||||
- name: Restore Python wheel cache
|
||||
if: inputs.lang == 'python'
|
||||
id: pyenv-cache
|
||||
uses: actions/cache/restore@v4
|
||||
uses: actions/cache/restore@v6
|
||||
with:
|
||||
path: |
|
||||
~/Library/Caches/pip
|
||||
@@ -251,24 +251,19 @@ jobs:
|
||||
BACKEND=${{ inputs.backend }} BUILD_TYPE=${{ inputs.build-type }} USE_PIP=${{ inputs.use-pip }} make build-darwin-${{ inputs.lang }}-backend
|
||||
|
||||
- name: ccache stats
|
||||
if: inputs.backend == 'llama-cpp' || inputs.backend == 'llama-cpp-localai-paged'
|
||||
if: inputs.backend == 'llama-cpp'
|
||||
run: ccache -s
|
||||
|
||||
# Only stock llama-cpp persists the ccache: both backends share the same
|
||||
# ccache-llama-<arch>-<version>-<run_id> key, so the paged job restores from
|
||||
# the shared prefix (warm) but must NOT also save under the identical key in
|
||||
# the same run (it would collide). The shared upstream TUs stay warm via the
|
||||
# stock save; the paged-only patched TUs are a small recompile.
|
||||
- name: Save ccache
|
||||
if: inputs.backend == 'llama-cpp' && github.event_name != 'pull_request'
|
||||
uses: actions/cache/save@v4
|
||||
uses: actions/cache/save@v6
|
||||
with:
|
||||
path: ~/Library/Caches/ccache
|
||||
key: ccache-llama-${{ runner.arch }}-${{ steps.llama-version.outputs.version }}-${{ github.run_id }}
|
||||
|
||||
- name: Save Python wheel cache
|
||||
if: inputs.lang == 'python' && github.event_name != 'pull_request' && steps.pyenv-cache.outputs.cache-hit != 'true'
|
||||
uses: actions/cache/save@v4
|
||||
uses: actions/cache/save@v6
|
||||
with:
|
||||
path: |
|
||||
~/Library/Caches/pip
|
||||
|
||||
165
.github/workflows/backend_pr.yml
vendored
165
.github/workflows/backend_pr.yml
vendored
@@ -11,16 +11,30 @@ jobs:
|
||||
generate-matrix:
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
matrix-singlearch: ${{ steps.set-matrix.outputs['matrix-singlearch'] }}
|
||||
matrix-multiarch: ${{ steps.set-matrix.outputs['matrix-multiarch'] }}
|
||||
matrix-darwin: ${{ steps.set-matrix.outputs['matrix-darwin'] }}
|
||||
merge-matrix-multiarch: ${{ steps.set-matrix.outputs['merge-matrix-multiarch'] }}
|
||||
merge-matrix-singlearch: ${{ steps.set-matrix.outputs['merge-matrix-singlearch'] }}
|
||||
has-backends-singlearch: ${{ steps.set-matrix.outputs['has-backends-singlearch'] }}
|
||||
has-backends-multiarch: ${{ steps.set-matrix.outputs['has-backends-multiarch'] }}
|
||||
has-backends-darwin: ${{ steps.set-matrix.outputs['has-backends-darwin'] }}
|
||||
has-merges-multiarch: ${{ steps.set-matrix.outputs['has-merges-multiarch'] }}
|
||||
has-merges-singlearch: ${{ steps.set-matrix.outputs['has-merges-singlearch'] }}
|
||||
# Single-arch backends are sharded across SINGLEARCH_SHARDS matrix jobs to
|
||||
# stay under GitHub's 256-jobs-per-matrix limit (see changed-backends.js).
|
||||
matrix-singlearch-1: ${{ steps.set-matrix.outputs['matrix-singlearch-1'] }}
|
||||
merge-matrix-singlearch-1: ${{ steps.set-matrix.outputs['merge-matrix-singlearch-1'] }}
|
||||
has-backends-singlearch-1: ${{ steps.set-matrix.outputs['has-backends-singlearch-1'] }}
|
||||
has-merges-singlearch-1: ${{ steps.set-matrix.outputs['has-merges-singlearch-1'] }}
|
||||
matrix-singlearch-2: ${{ steps.set-matrix.outputs['matrix-singlearch-2'] }}
|
||||
merge-matrix-singlearch-2: ${{ steps.set-matrix.outputs['merge-matrix-singlearch-2'] }}
|
||||
has-backends-singlearch-2: ${{ steps.set-matrix.outputs['has-backends-singlearch-2'] }}
|
||||
has-merges-singlearch-2: ${{ steps.set-matrix.outputs['has-merges-singlearch-2'] }}
|
||||
matrix-singlearch-3: ${{ steps.set-matrix.outputs['matrix-singlearch-3'] }}
|
||||
merge-matrix-singlearch-3: ${{ steps.set-matrix.outputs['merge-matrix-singlearch-3'] }}
|
||||
has-backends-singlearch-3: ${{ steps.set-matrix.outputs['has-backends-singlearch-3'] }}
|
||||
has-merges-singlearch-3: ${{ steps.set-matrix.outputs['has-merges-singlearch-3'] }}
|
||||
matrix-singlearch-4: ${{ steps.set-matrix.outputs['matrix-singlearch-4'] }}
|
||||
merge-matrix-singlearch-4: ${{ steps.set-matrix.outputs['merge-matrix-singlearch-4'] }}
|
||||
has-backends-singlearch-4: ${{ steps.set-matrix.outputs['has-backends-singlearch-4'] }}
|
||||
has-merges-singlearch-4: ${{ steps.set-matrix.outputs['has-merges-singlearch-4'] }}
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v7
|
||||
@@ -71,10 +85,10 @@ jobs:
|
||||
fail-fast: true
|
||||
max-parallel: 8
|
||||
matrix: ${{ fromJson(needs.generate-matrix.outputs['matrix-multiarch']) }}
|
||||
backend-jobs-singlearch:
|
||||
backend-jobs-singlearch-1:
|
||||
needs: generate-matrix
|
||||
if: needs.generate-matrix.outputs['has-backends-singlearch-1'] == 'true'
|
||||
uses: ./.github/workflows/backend_build.yml
|
||||
if: needs.generate-matrix.outputs['has-backends-singlearch'] == 'true'
|
||||
with:
|
||||
tag-latest: ${{ matrix.tag-latest }}
|
||||
tag-suffix: ${{ matrix.tag-suffix }}
|
||||
@@ -98,7 +112,94 @@ jobs:
|
||||
strategy:
|
||||
fail-fast: true
|
||||
max-parallel: 8
|
||||
matrix: ${{ fromJson(needs.generate-matrix.outputs['matrix-singlearch']) }}
|
||||
matrix: ${{ fromJson(needs.generate-matrix.outputs['matrix-singlearch-1']) }}
|
||||
|
||||
backend-jobs-singlearch-2:
|
||||
needs: generate-matrix
|
||||
if: needs.generate-matrix.outputs['has-backends-singlearch-2'] == 'true'
|
||||
uses: ./.github/workflows/backend_build.yml
|
||||
with:
|
||||
tag-latest: ${{ matrix.tag-latest }}
|
||||
tag-suffix: ${{ matrix.tag-suffix }}
|
||||
build-type: ${{ matrix.build-type }}
|
||||
cuda-major-version: ${{ matrix.cuda-major-version }}
|
||||
cuda-minor-version: ${{ matrix.cuda-minor-version }}
|
||||
platforms: ${{ matrix.platforms }}
|
||||
platform-tag: ${{ matrix.platform-tag || '' }}
|
||||
runs-on: ${{ matrix.runs-on }}
|
||||
builder-base-image: ${{ matrix.builder-base-image || '' }}
|
||||
base-image: ${{ matrix.base-image }}
|
||||
backend: ${{ matrix.backend }}
|
||||
dockerfile: ${{ matrix.dockerfile }}
|
||||
skip-drivers: ${{ matrix.skip-drivers }}
|
||||
context: ${{ matrix.context }}
|
||||
ubuntu-version: ${{ matrix.ubuntu-version }}
|
||||
amdgpu-targets: ${{ matrix.amdgpu-targets || 'gfx908,gfx90a,gfx942,gfx950,gfx1030,gfx1100,gfx1101,gfx1102,gfx1151,gfx1200,gfx1201' }}
|
||||
secrets:
|
||||
quayUsername: ${{ secrets.LOCALAI_REGISTRY_USERNAME }}
|
||||
quayPassword: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }}
|
||||
strategy:
|
||||
fail-fast: true
|
||||
max-parallel: 8
|
||||
matrix: ${{ fromJson(needs.generate-matrix.outputs['matrix-singlearch-2']) }}
|
||||
|
||||
backend-jobs-singlearch-3:
|
||||
needs: generate-matrix
|
||||
if: needs.generate-matrix.outputs['has-backends-singlearch-3'] == 'true'
|
||||
uses: ./.github/workflows/backend_build.yml
|
||||
with:
|
||||
tag-latest: ${{ matrix.tag-latest }}
|
||||
tag-suffix: ${{ matrix.tag-suffix }}
|
||||
build-type: ${{ matrix.build-type }}
|
||||
cuda-major-version: ${{ matrix.cuda-major-version }}
|
||||
cuda-minor-version: ${{ matrix.cuda-minor-version }}
|
||||
platforms: ${{ matrix.platforms }}
|
||||
platform-tag: ${{ matrix.platform-tag || '' }}
|
||||
runs-on: ${{ matrix.runs-on }}
|
||||
builder-base-image: ${{ matrix.builder-base-image || '' }}
|
||||
base-image: ${{ matrix.base-image }}
|
||||
backend: ${{ matrix.backend }}
|
||||
dockerfile: ${{ matrix.dockerfile }}
|
||||
skip-drivers: ${{ matrix.skip-drivers }}
|
||||
context: ${{ matrix.context }}
|
||||
ubuntu-version: ${{ matrix.ubuntu-version }}
|
||||
amdgpu-targets: ${{ matrix.amdgpu-targets || 'gfx908,gfx90a,gfx942,gfx950,gfx1030,gfx1100,gfx1101,gfx1102,gfx1151,gfx1200,gfx1201' }}
|
||||
secrets:
|
||||
quayUsername: ${{ secrets.LOCALAI_REGISTRY_USERNAME }}
|
||||
quayPassword: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }}
|
||||
strategy:
|
||||
fail-fast: true
|
||||
max-parallel: 8
|
||||
matrix: ${{ fromJson(needs.generate-matrix.outputs['matrix-singlearch-3']) }}
|
||||
|
||||
backend-jobs-singlearch-4:
|
||||
needs: generate-matrix
|
||||
if: needs.generate-matrix.outputs['has-backends-singlearch-4'] == 'true'
|
||||
uses: ./.github/workflows/backend_build.yml
|
||||
with:
|
||||
tag-latest: ${{ matrix.tag-latest }}
|
||||
tag-suffix: ${{ matrix.tag-suffix }}
|
||||
build-type: ${{ matrix.build-type }}
|
||||
cuda-major-version: ${{ matrix.cuda-major-version }}
|
||||
cuda-minor-version: ${{ matrix.cuda-minor-version }}
|
||||
platforms: ${{ matrix.platforms }}
|
||||
platform-tag: ${{ matrix.platform-tag || '' }}
|
||||
runs-on: ${{ matrix.runs-on }}
|
||||
builder-base-image: ${{ matrix.builder-base-image || '' }}
|
||||
base-image: ${{ matrix.base-image }}
|
||||
backend: ${{ matrix.backend }}
|
||||
dockerfile: ${{ matrix.dockerfile }}
|
||||
skip-drivers: ${{ matrix.skip-drivers }}
|
||||
context: ${{ matrix.context }}
|
||||
ubuntu-version: ${{ matrix.ubuntu-version }}
|
||||
amdgpu-targets: ${{ matrix.amdgpu-targets || 'gfx908,gfx90a,gfx942,gfx950,gfx1030,gfx1100,gfx1101,gfx1102,gfx1151,gfx1200,gfx1201' }}
|
||||
secrets:
|
||||
quayUsername: ${{ secrets.LOCALAI_REGISTRY_USERNAME }}
|
||||
quayPassword: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }}
|
||||
strategy:
|
||||
fail-fast: true
|
||||
max-parallel: 8
|
||||
matrix: ${{ fromJson(needs.generate-matrix.outputs['matrix-singlearch-4']) }}
|
||||
backend-merge-jobs-multiarch:
|
||||
needs: [generate-matrix, backend-jobs-multiarch]
|
||||
# backend_merge.yml's push-side steps are all gated on
|
||||
@@ -118,9 +219,9 @@ jobs:
|
||||
fail-fast: false
|
||||
matrix: ${{ fromJson(needs.generate-matrix.outputs['merge-matrix-multiarch']) }}
|
||||
|
||||
backend-merge-jobs-singlearch:
|
||||
needs: [generate-matrix, backend-jobs-singlearch]
|
||||
if: ${{ !cancelled() && github.event_name != 'pull_request' && needs.generate-matrix.outputs['has-merges-singlearch'] == 'true' }}
|
||||
backend-merge-jobs-singlearch-1:
|
||||
needs: [generate-matrix, backend-jobs-singlearch-1]
|
||||
if: ${{ !cancelled() && github.event_name != 'pull_request' && needs.generate-matrix.outputs['has-merges-singlearch-1'] == 'true' }}
|
||||
uses: ./.github/workflows/backend_merge.yml
|
||||
with:
|
||||
tag-latest: ${{ matrix.tag-latest }}
|
||||
@@ -130,7 +231,49 @@ jobs:
|
||||
quayPassword: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix: ${{ fromJson(needs.generate-matrix.outputs['merge-matrix-singlearch']) }}
|
||||
matrix: ${{ fromJson(needs.generate-matrix.outputs['merge-matrix-singlearch-1']) }}
|
||||
|
||||
backend-merge-jobs-singlearch-2:
|
||||
needs: [generate-matrix, backend-jobs-singlearch-2]
|
||||
if: ${{ !cancelled() && github.event_name != 'pull_request' && needs.generate-matrix.outputs['has-merges-singlearch-2'] == 'true' }}
|
||||
uses: ./.github/workflows/backend_merge.yml
|
||||
with:
|
||||
tag-latest: ${{ matrix.tag-latest }}
|
||||
tag-suffix: ${{ matrix.tag-suffix }}
|
||||
secrets:
|
||||
quayUsername: ${{ secrets.LOCALAI_REGISTRY_USERNAME }}
|
||||
quayPassword: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix: ${{ fromJson(needs.generate-matrix.outputs['merge-matrix-singlearch-2']) }}
|
||||
|
||||
backend-merge-jobs-singlearch-3:
|
||||
needs: [generate-matrix, backend-jobs-singlearch-3]
|
||||
if: ${{ !cancelled() && github.event_name != 'pull_request' && needs.generate-matrix.outputs['has-merges-singlearch-3'] == 'true' }}
|
||||
uses: ./.github/workflows/backend_merge.yml
|
||||
with:
|
||||
tag-latest: ${{ matrix.tag-latest }}
|
||||
tag-suffix: ${{ matrix.tag-suffix }}
|
||||
secrets:
|
||||
quayUsername: ${{ secrets.LOCALAI_REGISTRY_USERNAME }}
|
||||
quayPassword: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix: ${{ fromJson(needs.generate-matrix.outputs['merge-matrix-singlearch-3']) }}
|
||||
|
||||
backend-merge-jobs-singlearch-4:
|
||||
needs: [generate-matrix, backend-jobs-singlearch-4]
|
||||
if: ${{ !cancelled() && github.event_name != 'pull_request' && needs.generate-matrix.outputs['has-merges-singlearch-4'] == 'true' }}
|
||||
uses: ./.github/workflows/backend_merge.yml
|
||||
with:
|
||||
tag-latest: ${{ matrix.tag-latest }}
|
||||
tag-suffix: ${{ matrix.tag-suffix }}
|
||||
secrets:
|
||||
quayUsername: ${{ secrets.LOCALAI_REGISTRY_USERNAME }}
|
||||
quayPassword: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix: ${{ fromJson(needs.generate-matrix.outputs['merge-matrix-singlearch-4']) }}
|
||||
backend-jobs-darwin:
|
||||
needs: generate-matrix
|
||||
uses: ./.github/workflows/backend_build_darwin.yml
|
||||
|
||||
37
.github/workflows/bump_deps.yaml
vendored
37
.github/workflows/bump_deps.yaml
vendored
@@ -9,23 +9,6 @@ jobs:
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
# NOTE: there is intentionally NO entry for the llama-cpp-localai-paged
|
||||
# backend. It carries a vendored paged-attention patch series
|
||||
# (backend/cpp/llama-cpp-localai-paged/patches/paged/) hand-verified bit-exact against
|
||||
# ONE specific llama.cpp tip; a naive nightly bump would move the tip out
|
||||
# from under the patches and break `git apply` at build time. Its pin is
|
||||
# therefore decoupled (its own LLAMA_VERSION in
|
||||
# backend/cpp/llama-cpp-localai-paged/Makefile) and advanced ONLY by the
|
||||
# manual PIN_SYNC process. Do not add it here. (turboquant CAN be
|
||||
# auto-bumped below because its fork branch carries the patches.)
|
||||
#
|
||||
# Excluding it from the auto-bumper removed the early warning of upstream
|
||||
# drift; that signal is restored separately by the dedicated canary
|
||||
# .github/workflows/llama-cpp-paged-canary.yml, which weekly applies +
|
||||
# compiles the paged series against the latest llama.cpp tip and goes red
|
||||
# when upstream breaks it (prompting a PIN_SYNC). The canary is
|
||||
# signal-only - it never opens a bump PR and never moves the pin - so
|
||||
# this dep-bump workflow and its PRs stay green regardless.
|
||||
include:
|
||||
- repository: "ggml-org/llama.cpp"
|
||||
variable: "LLAMA_VERSION"
|
||||
@@ -39,10 +22,18 @@ jobs:
|
||||
variable: "TURBOQUANT_VERSION"
|
||||
branch: "feature/turboquant-kv-cache"
|
||||
file: "backend/cpp/turboquant/Makefile"
|
||||
- repository: "PrismML-Eng/llama.cpp"
|
||||
variable: "BONSAI_VERSION"
|
||||
branch: "prism"
|
||||
file: "backend/cpp/bonsai/Makefile"
|
||||
- repository: "antirez/ds4"
|
||||
variable: "DS4_VERSION"
|
||||
branch: "main"
|
||||
file: "backend/cpp/ds4/Makefile"
|
||||
- repository: "meituan-longcat/LongCat-Video"
|
||||
variable: "LONGCAT_VIDEO_VERSION"
|
||||
branch: "main"
|
||||
file: "backend/python/longcat-video/Makefile"
|
||||
- repository: "localai-org/privacy-filter.cpp"
|
||||
variable: "PRIVACY_FILTER_VERSION"
|
||||
branch: "master"
|
||||
@@ -59,11 +50,15 @@ jobs:
|
||||
variable: "PARAKEET_VERSION"
|
||||
branch: "master"
|
||||
file: "backend/go/parakeet-cpp/Makefile"
|
||||
- repository: "mudler/ced.cpp"
|
||||
variable: "CED_VERSION"
|
||||
- repository: "localai-org/moss-transcribe.cpp"
|
||||
variable: "MOSS_VERSION"
|
||||
branch: "master"
|
||||
file: "backend/go/moss-transcribe-cpp/Makefile"
|
||||
- repository: "localai-org/ced.cpp"
|
||||
variable: "CED_VERSION"
|
||||
branch: "main"
|
||||
file: "backend/go/ced/Makefile"
|
||||
- repository: "mudler/voice-detect.cpp"
|
||||
- repository: "localai-org/voice-detect.cpp"
|
||||
variable: "VOICEDETECT_VERSION"
|
||||
branch: "master"
|
||||
file: "backend/go/voice-detect/Makefile"
|
||||
@@ -95,7 +90,7 @@ jobs:
|
||||
variable: "SAM3_VERSION"
|
||||
branch: "main"
|
||||
file: "backend/go/sam3-cpp/Makefile"
|
||||
- repository: "mudler/rf-detr.cpp"
|
||||
- repository: "localai-org/rf-detr.cpp"
|
||||
variable: "RFDETR_VERSION"
|
||||
branch: "main"
|
||||
file: "backend/go/rfdetr-cpp/Makefile"
|
||||
|
||||
54
.github/workflows/gallery_variant_proposals.yaml
vendored
Normal file
54
.github/workflows/gallery_variant_proposals.yaml
vendored
Normal file
@@ -0,0 +1,54 @@
|
||||
name: Propose gallery variant groupings
|
||||
on:
|
||||
schedule:
|
||||
- cron: 0 4 * * 1
|
||||
workflow_dispatch:
|
||||
jobs:
|
||||
variant_proposals:
|
||||
if: github.repository == 'mudler/LocalAI'
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
- uses: actions/setup-go@v5
|
||||
with:
|
||||
go-version-file: go.mod
|
||||
cache: false
|
||||
|
||||
# The heuristics are the risky part of this job. A regression in them
|
||||
# produces confident, wrong proposals, which is worse than no job at all.
|
||||
- name: Test the proposer
|
||||
run: go test ./.github/ci/variantproposals/
|
||||
|
||||
- name: Propose groupings 🔧
|
||||
id: propose
|
||||
run: |
|
||||
rm -f /tmp/variant-proposals-body.md
|
||||
go run ./.github/ci/variantproposals \
|
||||
-index gallery/index.yaml \
|
||||
-ledger gallery/variant-exclusions.yaml \
|
||||
-body-out /tmp/variant-proposals-body.md \
|
||||
-apply
|
||||
if [ -s /tmp/variant-proposals-body.md ]; then
|
||||
echo "have_proposals=true" >> "$GITHUB_OUTPUT"
|
||||
{
|
||||
echo 'body<<VARIANT_PROPOSAL_BODY_EOF'
|
||||
cat /tmp/variant-proposals-body.md
|
||||
echo VARIANT_PROPOSAL_BODY_EOF
|
||||
} >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "have_proposals=false" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
# No body file means the proposer found nothing. Opening an empty pull
|
||||
# request every run is how a proposal job gets muted by its reviewers.
|
||||
- name: Create Pull Request
|
||||
if: steps.propose.outputs.have_proposals == 'true'
|
||||
uses: peter-evans/create-pull-request@v8
|
||||
with:
|
||||
token: ${{ secrets.UPDATE_BOT_TOKEN }}
|
||||
push-to-fork: ci-forks/LocalAI
|
||||
commit-message: 'chore(model-gallery): propose variant groupings'
|
||||
title: 'chore(model-gallery): propose variant groupings for review'
|
||||
branch: "propose/variant-groupings"
|
||||
body: ${{ steps.propose.outputs.body }}
|
||||
signoff: true
|
||||
20
.github/workflows/lint.yml
vendored
20
.github/workflows/lint.yml
vendored
@@ -46,3 +46,23 @@ jobs:
|
||||
touch core/http/react-ui/dist/index.html
|
||||
- name: lint
|
||||
run: make lint
|
||||
|
||||
build-scripts:
|
||||
# The image packaging scripts encode invariants that only surface inside a
|
||||
# container build (a missing transitive dep, a partial cuDNN family). Their
|
||||
# shell tests need nothing but bash + gcc + ldd, so run them on every PR
|
||||
# rather than waiting on a multi-GB cross-arch backend image build.
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
- name: run packaging script tests
|
||||
run: make test-build-scripts
|
||||
|
||||
# The backend matrix path filter fails silently: a miss emits an empty
|
||||
# matrix, every job goes green, and the change reaches no image (#10946).
|
||||
# Its tests need only node, so they ride along with this job.
|
||||
- uses: actions/setup-node@v4
|
||||
with:
|
||||
node-version: '20'
|
||||
- name: run CI script tests
|
||||
run: make test-ci-scripts
|
||||
|
||||
179
.github/workflows/llama-cpp-paged-canary.yml
vendored
179
.github/workflows/llama-cpp-paged-canary.yml
vendored
@@ -1,179 +0,0 @@
|
||||
name: 'llama.cpp paged patches: upstream canary'
|
||||
|
||||
# EARLY-WARNING CANARY for the vendored paged-attention patch series
|
||||
# (backend/cpp/llama-cpp-localai-paged/patches/paged/0001-0030).
|
||||
#
|
||||
# WHY THIS EXISTS
|
||||
# The paged backend (backend/cpp/llama-cpp-localai-paged) pins its OWN verified
|
||||
# llama.cpp tip (LLAMA_VERSION in backend/cpp/llama-cpp-localai-paged/Makefile)
|
||||
# and is intentionally EXCLUDED from the nightly auto-bumper
|
||||
# (.github/workflows/bump_deps.yaml), so a naive upstream bump can never silently
|
||||
# break the shipped build. The cost of that safety: nobody finds out when
|
||||
# upstream DRIFTS past the patches. This canary restores that signal WITHOUT
|
||||
# touching the shipped pin - weekly it tries the patch series + a real compile
|
||||
# against the LATEST llama.cpp master tip and goes red the moment upstream breaks
|
||||
# the patches.
|
||||
#
|
||||
# RED HERE means: time to run a PIN_SYNC (rebase the patches onto the new tip,
|
||||
# pass the bit-exact gate on the GPU, re-export the .patch files, THEN advance
|
||||
# the pin in backend/cpp/llama-cpp-localai-paged/Makefile). See the backend README
|
||||
# section 7 (Pin + maintenance policy):
|
||||
# backend/cpp/llama-cpp-localai-paged/README.md.
|
||||
#
|
||||
# SIGNAL-ONLY: this workflow moves no pinned version, ships nothing, and is fully
|
||||
# decoupled from bump_deps - so the main dep-bump PR stays green regardless. A
|
||||
# green run means "the paged series still applies and compiles on upstream HEAD";
|
||||
# a red run means "upstream moved - schedule a pin-sync".
|
||||
|
||||
on:
|
||||
schedule:
|
||||
# Weekly (Mondays 06:00 UTC), mirroring the weekly DEPS_REFRESH / bump_deps
|
||||
# cadence. Offset from bump_deps' nightly 20:00 so the two never pile up.
|
||||
- cron: '0 6 * * 1'
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: llama-cpp-paged-canary
|
||||
cancel-in-progress: false
|
||||
|
||||
env:
|
||||
# Upstream source of truth - the same repo/branch bump_deps tracks for the
|
||||
# stock llama-cpp pin.
|
||||
LLAMA_UPSTREAM: 'https://github.com/ggml-org/llama.cpp'
|
||||
|
||||
jobs:
|
||||
apply-check:
|
||||
# Cheap, fast, toolchain-free early warning: does the series still APPLY to
|
||||
# the latest upstream tip? A patch no longer applying is by far the most
|
||||
# common way upstream breaks a vendored series, so this runs first, is
|
||||
# reliable on a free runner, and feeds the resolved tip to the compile job.
|
||||
if: github.repository == 'mudler/LocalAI'
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 20
|
||||
outputs:
|
||||
tip: ${{ steps.resolve.outputs.tip }}
|
||||
steps:
|
||||
- name: Checkout LocalAI
|
||||
uses: actions/checkout@v7
|
||||
|
||||
- name: Resolve latest llama.cpp master tip
|
||||
id: resolve
|
||||
run: |
|
||||
tip="$(git ls-remote "$LLAMA_UPSTREAM" refs/heads/master | cut -f1)"
|
||||
if [ -z "$tip" ]; then
|
||||
echo "::error::could not resolve llama.cpp master tip from $LLAMA_UPSTREAM"
|
||||
exit 1
|
||||
fi
|
||||
pin="$(grep -m1 'LLAMA_VERSION?=' backend/cpp/llama-cpp-localai-paged/Makefile | cut -d= -f2)"
|
||||
echo "latest llama.cpp master tip: $tip"
|
||||
echo "shipped paged pin: $pin"
|
||||
echo "tip=$tip" >> "$GITHUB_OUTPUT"
|
||||
{
|
||||
echo "## llama.cpp paged canary"
|
||||
echo ""
|
||||
echo "- upstream master tip: \`$tip\`"
|
||||
echo "- shipped paged pin: \`$pin\`"
|
||||
} >> "$GITHUB_STEP_SUMMARY"
|
||||
|
||||
- name: Checkout llama.cpp at latest tip (shallow)
|
||||
run: |
|
||||
mkdir -p /tmp/llama.cpp
|
||||
cd /tmp/llama.cpp
|
||||
git init -q
|
||||
git remote add origin "$LLAMA_UPSTREAM"
|
||||
git fetch -q --depth 1 origin "${{ steps.resolve.outputs.tip }}"
|
||||
git checkout -q FETCH_HEAD
|
||||
git log --oneline -1
|
||||
|
||||
- name: Apply paged patch series (build's git-apply method)
|
||||
run: |
|
||||
bash .github/scripts/paged-canary-apply.sh \
|
||||
/tmp/llama.cpp \
|
||||
"$PWD/backend/cpp/llama-cpp-localai-paged/patches"
|
||||
echo "- apply: full paged series applies to the upstream tip :white_check_mark:" >> "$GITHUB_STEP_SUMMARY"
|
||||
|
||||
compile:
|
||||
# Proves the patches still COMPILE against the latest tip, using the SAME
|
||||
# toolchain + build target the shipped paged backend uses (the
|
||||
# base-grpc-cuda-12 builder base + the Makefile `grpc-server` cublas target),
|
||||
# so a failure means upstream drift, not toolchain noise. CUDA is compiled
|
||||
# (nvcc; no GPU required) because most of the paged series is CUDA kernels.
|
||||
# Runs only if the apply check passed, on the exact tip it validated.
|
||||
#
|
||||
# If a full CUDA compile on the hosted runner ever proves too heavy/flaky,
|
||||
# switch `runs-on` to 'bigger-runner' (the runner class the real paged CUDA
|
||||
# build uses), or drop to a CPU build (BUILD_TYPE='') which still compiles
|
||||
# all host + CPU paged code, leaving CUDA-kernel coverage to the apply check
|
||||
# plus the manual PIN_SYNC GPU gate.
|
||||
needs: apply-check
|
||||
if: github.repository == 'mudler/LocalAI'
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 180
|
||||
steps:
|
||||
- name: Checkout LocalAI
|
||||
uses: actions/checkout@v7
|
||||
|
||||
- name: Free disk space
|
||||
uses: ./.github/actions/free-disk-space
|
||||
with:
|
||||
mode: hosted
|
||||
|
||||
- name: Login to Quay.io
|
||||
uses: docker/login-action@v4
|
||||
with:
|
||||
registry: quay.io
|
||||
username: ${{ secrets.LOCALAI_REGISTRY_USERNAME }}
|
||||
password: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }}
|
||||
|
||||
- name: Compile paged backend against latest tip (cublas)
|
||||
env:
|
||||
TIP: ${{ needs.apply-check.outputs.tip }}
|
||||
BUILDER_BASE_IMAGE: 'quay.io/go-skynet/ci-cache:base-grpc-cuda-12-amd64'
|
||||
run: |
|
||||
docker run --rm \
|
||||
-v "$PWD":/LocalAI -w /LocalAI \
|
||||
-e TIP -e LLAMA_UPSTREAM \
|
||||
"$BUILDER_BASE_IMAGE" bash -euxo pipefail -c '
|
||||
# Mirror the Dockerfile: gRPC lives at /opt/grpc in the base image;
|
||||
# copy it to the prefix CMake find_package expects.
|
||||
cp -a /opt/grpc/. /usr/local/
|
||||
|
||||
# Pre-populate the llama.cpp checkout at the latest tip with the
|
||||
# paged series applied via the tolerant canary apply. Because
|
||||
# backend/cpp/llama-cpp/llama.cpp now exists, the stock Makefile's
|
||||
# llama.cpp target (clone + base-patch apply) is skipped and the
|
||||
# now patch-free prepare.sh only copies the grpc-server sources -
|
||||
# so we drive the REAL grpc-server build path on top of our paged
|
||||
# apply. The stock llama-cpp backend no longer carries the paged
|
||||
# series (it lives in backend/cpp/llama-cpp-localai-paged/patches/
|
||||
# paged); we build it here in the stock dir only because that is
|
||||
# where the shared build infra (Makefile / grpc-server.cpp /
|
||||
# CMakeLists.txt / prepare.sh) lives.
|
||||
cd backend/cpp/llama-cpp/
|
||||
mkdir -p llama.cpp
|
||||
cd llama.cpp
|
||||
git init -q
|
||||
git remote add origin "$LLAMA_UPSTREAM"
|
||||
git fetch -q --depth 1 origin "$TIP"
|
||||
git checkout -q FETCH_HEAD
|
||||
cd /LocalAI
|
||||
bash .github/scripts/paged-canary-apply.sh \
|
||||
backend/cpp/llama-cpp/llama.cpp \
|
||||
"$PWD/backend/cpp/llama-cpp-localai-paged/patches"
|
||||
|
||||
# Cheapest real CUDA build that proves the patches compile: one
|
||||
# CUDA arch, cublas. CMAKE_ARGS is passed via the environment (not
|
||||
# as a make arg) so the Makefile += flags are still appended,
|
||||
# exactly like .docker/llama-cpp-localai-paged-compile.sh. The paged
|
||||
# series is already applied to the checkout above, so the stock
|
||||
# build just compiles the patched tree.
|
||||
cd backend/cpp/llama-cpp/
|
||||
BUILD_TYPE=cublas \
|
||||
CMAKE_ARGS="-DCMAKE_CUDA_ARCHITECTURES=80" \
|
||||
make grpc-server
|
||||
test -x grpc-server
|
||||
'
|
||||
echo "- compile: paged series builds (cublas) against the upstream tip :white_check_mark:" >> "$GITHUB_STEP_SUMMARY"
|
||||
69
.github/workflows/realtime-conformance.yml
vendored
Normal file
69
.github/workflows/realtime-conformance.yml
vendored
Normal file
@@ -0,0 +1,69 @@
|
||||
---
|
||||
name: 'realtime-conformance'
|
||||
|
||||
# Verifies the realtime state-machine implementations conform to their formal
|
||||
# designs (docs/design/realtime-state-machines.md, formal-verification/). BOTH
|
||||
# layers are enforced and the gate is fail-closed: the Go conformance layer
|
||||
# (respcoord + turncoord transition/rapid tests under -race) AND the FizzBee model check of
|
||||
# the authoritative specs. FizzBee is pinned + checksum-verified
|
||||
# (formal-verification/fizzbee.sha256), so a failed install fails the job rather
|
||||
# than silently skipping verification.
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
paths:
|
||||
- 'core/http/endpoints/openai/coordinator/**'
|
||||
- 'core/http/endpoints/openai/respcoord/**'
|
||||
- 'core/http/endpoints/openai/turncoord/**'
|
||||
- 'core/http/endpoints/openai/conncoord/**'
|
||||
- 'core/http/endpoints/openai/compactcoord/**'
|
||||
- 'core/http/endpoints/openai/ttscoord/**'
|
||||
- 'formal-verification/**'
|
||||
- 'scripts/realtime-conformance.sh'
|
||||
- 'scripts/install-fizzbee.sh'
|
||||
- '.github/workflows/realtime-conformance.yml'
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
paths:
|
||||
- 'core/http/endpoints/openai/coordinator/**'
|
||||
- 'core/http/endpoints/openai/respcoord/**'
|
||||
- 'core/http/endpoints/openai/turncoord/**'
|
||||
- 'core/http/endpoints/openai/conncoord/**'
|
||||
- 'core/http/endpoints/openai/compactcoord/**'
|
||||
- 'core/http/endpoints/openai/ttscoord/**'
|
||||
- 'formal-verification/**'
|
||||
- 'scripts/realtime-conformance.sh'
|
||||
|
||||
concurrency:
|
||||
group: realtime-conformance-${{ github.event.pull_request.number || github.sha }}-${{ github.repository }}
|
||||
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
|
||||
|
||||
jobs:
|
||||
conformance:
|
||||
runs-on: ubuntu-latest
|
||||
strategy:
|
||||
matrix:
|
||||
go-version: ['1.26.x']
|
||||
steps:
|
||||
- name: Clone
|
||||
uses: actions/checkout@v7
|
||||
- name: Setup Go ${{ matrix.go-version }}
|
||||
uses: actions/setup-go@v5
|
||||
with:
|
||||
go-version: ${{ matrix.go-version }}
|
||||
cache: false
|
||||
- name: Cache FizzBee
|
||||
uses: actions/cache@v6
|
||||
with:
|
||||
path: .tools/fizzbee
|
||||
key: fizzbee-v0.5.2-${{ runner.os }}-${{ hashFiles('formal-verification/fizzbee.sha256') }}
|
||||
- name: Install FizzBee (pinned, checksum-verified)
|
||||
# No `|| true`: a failed/forged download must fail the job, not silently
|
||||
# drop the design verification. install-fizzbee.sh is a no-op if the
|
||||
# cached binary is already present and valid.
|
||||
run: ./scripts/install-fizzbee.sh
|
||||
- name: Run conformance gate (fail-closed)
|
||||
# No skip env: both the Go conformance and the FizzBee model check are
|
||||
# required. The gate auto-detects .tools/fizzbee/fizz.
|
||||
run: make test-realtime-conformance
|
||||
2
.github/workflows/stalebot.yml
vendored
2
.github/workflows/stalebot.yml
vendored
@@ -11,7 +11,7 @@ jobs:
|
||||
if: github.repository == 'mudler/LocalAI'
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/stale@eb5cf3af3ac0a1aa4c9c45633dd1ae542a27a899 # v9
|
||||
- uses: actions/stale@1e223db275d687790206a7acac4d1a11bd6fe629 # v9
|
||||
with:
|
||||
stale-issue-message: 'This issue is stale because it has been open 90 days with no activity. Remove stale label or comment or this will be closed in 5 days.'
|
||||
stale-pr-message: 'This PR is stale because it has been open 90 days with no activity. Remove stale label or comment or this will be closed in 10 days.'
|
||||
|
||||
2
.github/workflows/test-extra.yml
vendored
2
.github/workflows/test-extra.yml
vendored
@@ -587,7 +587,7 @@ jobs:
|
||||
with:
|
||||
go-version: '1.25.4'
|
||||
- name: Setup Node.js
|
||||
uses: actions/setup-node@v6
|
||||
uses: actions/setup-node@v7
|
||||
with:
|
||||
node-version: '22'
|
||||
- name: Build sherpa-onnx backend image and run realtime e2e tests
|
||||
|
||||
4
.github/workflows/test.yml
vendored
4
.github/workflows/test.yml
vendored
@@ -48,7 +48,7 @@ jobs:
|
||||
sudo apt-get update
|
||||
sudo apt-get install curl ffmpeg libopus-dev
|
||||
- name: Setup Node.js
|
||||
uses: actions/setup-node@v6
|
||||
uses: actions/setup-node@v7
|
||||
with:
|
||||
node-version: '22'
|
||||
- name: Build React UI
|
||||
@@ -100,7 +100,7 @@ jobs:
|
||||
brew install protobuf grpc make protoc-gen-go protoc-gen-go-grpc libomp llvm opus ffmpeg
|
||||
pip install --user --no-cache-dir grpcio-tools grpcio
|
||||
- name: Setup Node.js
|
||||
uses: actions/setup-node@v6
|
||||
uses: actions/setup-node@v7
|
||||
with:
|
||||
node-version: '22'
|
||||
- name: Build React UI
|
||||
|
||||
2
.github/workflows/tests-e2e.yml
vendored
2
.github/workflows/tests-e2e.yml
vendored
@@ -47,7 +47,7 @@ jobs:
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y build-essential libopus-dev
|
||||
- name: Setup Node.js
|
||||
uses: actions/setup-node@v6
|
||||
uses: actions/setup-node@v7
|
||||
with:
|
||||
node-version: '22'
|
||||
- name: Build React UI
|
||||
|
||||
2
.github/workflows/tests-ui-e2e.yml
vendored
2
.github/workflows/tests-ui-e2e.yml
vendored
@@ -34,7 +34,7 @@ jobs:
|
||||
go-version: ${{ matrix.go-version }}
|
||||
cache: false
|
||||
- name: Setup Node.js
|
||||
uses: actions/setup-node@v6
|
||||
uses: actions/setup-node@v7
|
||||
with:
|
||||
node-version: '22'
|
||||
- name: Setup Bun
|
||||
|
||||
25
.gitignore
vendored
25
.gitignore
vendored
@@ -9,15 +9,6 @@ prepare-sources
|
||||
/backend/cpp/llama-cpp/llama.cpp
|
||||
/backend/cpp/llama-*
|
||||
!backend/cpp/llama-cpp
|
||||
# llama-cpp-localai-paged is a tracked source dir (a thin wrapper Makefile over
|
||||
# backend/cpp/llama-cpp). Re-include it like llama-cpp above; its sibling
|
||||
# *-build dirs are still ignored by the /backend/cpp/llama-* rule, and its
|
||||
# in-dir build artifacts (binaries, package output, collected ggml .so set) are
|
||||
# re-ignored just below.
|
||||
!backend/cpp/llama-cpp-localai-paged
|
||||
/backend/cpp/llama-cpp-localai-paged/llama-cpp-localai-paged-*
|
||||
/backend/cpp/llama-cpp-localai-paged/package
|
||||
/backend/cpp/llama-cpp-localai-paged/ggml-shared-libs
|
||||
/backends
|
||||
/backend-images
|
||||
/result.yaml
|
||||
@@ -50,7 +41,12 @@ models/*
|
||||
test-models/
|
||||
test-dir/
|
||||
tests/e2e-aio/backends
|
||||
mock-backend
|
||||
# The mock backend binary built by `make build-mock-backend`. Anchored to its
|
||||
# full path: a bare `mock-backend` also matched the *directory* holding the
|
||||
# source, so git would not descend into it and adding a file there needed -f.
|
||||
# tests/e2e/mock-backend/.gitignore covers the same binary; kept here too so
|
||||
# the artifact stays ignored if that scoped file is ever removed.
|
||||
/tests/e2e/mock-backend/mock-backend
|
||||
|
||||
release/
|
||||
|
||||
@@ -106,3 +102,12 @@ core/http/react-ui/test-results/
|
||||
|
||||
# Local Apple signing material (never commit)
|
||||
.certs/
|
||||
|
||||
# Pinned dev tools (e.g. FizzBee for the realtime-conformance gate)
|
||||
.tools/
|
||||
|
||||
# FizzBee model-check artifacts: the parser emits <spec>.json next to each
|
||||
# .fizz and the checker writes run dirs under out/. Both are regenerated by
|
||||
# the realtime-conformance gate; only the .fizz sources are authoritative.
|
||||
formal-verification/*.json
|
||||
formal-verification/out/
|
||||
|
||||
6
.impeccable/live/config.json
Normal file
6
.impeccable/live/config.json
Normal file
@@ -0,0 +1,6 @@
|
||||
{
|
||||
"files": ["core/http/react-ui/index.html"],
|
||||
"insertBefore": "</body>",
|
||||
"commentSyntax": "html",
|
||||
"cspChecked": true
|
||||
}
|
||||
@@ -23,8 +23,6 @@ LocalAI follows the Linux kernel project's [guidelines for AI coding assistants]
|
||||
| [.agents/adding-backends.md](.agents/adding-backends.md) | Adding a new backend (Python, Go, or C++) — full step-by-step checklist, including importer integration (the `/import-model` dropdown is server-driven from `GET /backends/known`) |
|
||||
| [.agents/coding-style.md](.agents/coding-style.md) | Code style, editorconfig, logging, documentation conventions |
|
||||
| [.agents/llama-cpp-backend.md](.agents/llama-cpp-backend.md) | Working on the llama.cpp backend — architecture, updating, tool call parsing |
|
||||
| [.agents/llama-cpp-localai-paged-backend.md](.agents/llama-cpp-localai-paged-backend.md) | Working on the CUDA-only paged-attention llama.cpp variant (Qwen3.6 hybrid-SSM / Blackwell NVFP4 decode) - patchset scope, the bit-exact gate, the manual pin-sync + weekly canary, CUDA-only invariants, stock-stays-pure, Metal/SYCL/Vulkan follow-up scope |
|
||||
| [.agents/vllm-parity-methodology.md](.agents/vllm-parity-methodology.md) | The methodology for closing the vLLM decode-throughput gap in llama.cpp - bit-exact gating, profile-don't-assume, both-engine ground-truth, per-lever A/B discipline, recording rejected levers, multi-agent GPU orchestration |
|
||||
| [.agents/vllm-backend.md](.agents/vllm-backend.md) | Working on the vLLM / vLLM-omni backends — native parsers, ChatDelta, CPU build, libnuma packaging, backend hooks |
|
||||
| [.agents/sglang-backend.md](.agents/sglang-backend.md) | Working on the SGLang backend — `engine_args` validation against ServerArgs, speculative-decoding (EAGLE/EAGLE3/DFLASH/MTP) recipes, parser handling |
|
||||
| [.agents/ds4-backend.md](.agents/ds4-backend.md) | Working on the ds4 backend - DSML state machine, thinking modes, KV cache, Metal+CUDA matrix |
|
||||
@@ -39,12 +37,12 @@ LocalAI follows the Linux kernel project's [guidelines for AI coding assistants]
|
||||
|
||||
- **Git hooks & coverage gates**: Run `make install-hooks` once per clone so the pre-commit lint + coverage gates run. **Never bypass them with `git commit --no-verify`, and never lower a coverage baseline or widen a gate's tolerance to turn a red gate green** — the coverage ratchet only moves up. If a change drops coverage, add tests to raise it (e.g. render-smoke specs). See [.agents/building-and-testing.md](.agents/building-and-testing.md).
|
||||
- **Logging**: Use `github.com/mudler/xlog` (same API as slog)
|
||||
- **Paged llama.cpp backend**: `llama-cpp-localai-paged` is a CUDA-only variant that owns its own patch series + its own pinned llama.cpp (manual pin-sync, weekly canary); the stock `llama-cpp` backend stays patch-free. Read [.agents/llama-cpp-localai-paged-backend.md](.agents/llama-cpp-localai-paged-backend.md) before touching either, and [.agents/vllm-parity-methodology.md](.agents/vllm-parity-methodology.md) for the decode-parity methodology behind it.
|
||||
- **Go style**: Prefer `any` over `interface{}`
|
||||
- **Comments**: Explain *why*, not *what*
|
||||
- **Docs**: Update `docs/content/` when adding features or changing config
|
||||
- **Docs (docs-with-code rule)**: When you change user-facing behavior (API endpoints, CLI flags, config keys, or features), update the corresponding page under `docs/content/` in the SAME change, not as a follow-up. A user-facing change without a matching docs update is incomplete. See also the documentation conventions in [.agents/coding-style.md](.agents/coding-style.md).
|
||||
- **New API endpoints**: LocalAI advertises its capability surface in several independent places — swagger `@Tags`, `/api/instructions` registry, auth `RouteFeatureRegistry`, React UI `capabilities.js`, docs. Read [.agents/api-endpoints-and-auth.md](.agents/api-endpoints-and-auth.md) and follow its checklist — missing any surface means clients, admins, and the UI won't know the endpoint exists.
|
||||
- **Admin endpoints → MCP tool**: every admin endpoint that an admin would manage conversationally (install/list/edit/toggle/upgrade) MUST also be exposed as an MCP tool in `pkg/mcp/localaitools/`. The LocalAI Assistant chat modality and the standalone `local-ai mcp-server` consume that package; drift between REST and MCP is a real risk. Read [.agents/localai-assistant-mcp.md](.agents/localai-assistant-mcp.md) — the `TestToolHTTPRouteMappingComplete` test fails until you wire the new tool and update the route map.
|
||||
- **Build**: Inspect `Makefile` and `.github/workflows/` — ask the user before running long builds
|
||||
- **Backend OS coverage**: a new backend must target every OS it can build for, not just Linux. `.github/backend-matrix.yml` has two matrices — `include:` (Linux) and `includeDarwin:` (macOS / Apple Silicon). Most C/C++/GGML and many Python backends build on Darwin too — wire the `includeDarwin` entry + `backend/index.yaml` `metal:` entries, or say in the PR why an OS is unsupported. See the darwin checklist in [.agents/adding-backends.md](.agents/adding-backends.md).
|
||||
- **Gallery variant ranking**: a gallery entry can declare `variants` (alternative builds of the same weights), and LocalAI ranks the ones a host can run by engine preference first, size second. A new backend that should be preferred on some hardware must be listed in `engineNamePreferenceRules` in `pkg/system/capabilities.go`; the sibling `backendBuildTagPreferenceRules` speaks build tags rather than engine names, and using the wrong table matches nothing without erroring. See [.agents/adding-backends.md](.agents/adding-backends.md).
|
||||
- **UI**: The active UI is the React app in `core/http/react-ui/`. The older Alpine.js/HTML UI in `core/http/static/` is pending deprecation — all new UI work goes in the React UI
|
||||
|
||||
44
Dockerfile
44
Dockerfile
@@ -12,12 +12,16 @@ ARG APT_MIRROR
|
||||
ARG APT_PORTS_MIRROR
|
||||
ENV DEBIAN_FRONTEND=noninteractive
|
||||
|
||||
# hwdata ships /usr/share/hwdata/pci.ids. Without it, the ghw library we use
|
||||
# for hardware detection cannot resolve PCI vendor IDs and fails to enumerate
|
||||
# GPUs at all, so the image reports "No GPU detected" (see issue #10941).
|
||||
RUN --mount=type=bind,source=.docker/apt-mirror.sh,target=/usr/local/sbin/apt-mirror \
|
||||
APT_MIRROR="${APT_MIRROR}" APT_PORTS_MIRROR="${APT_PORTS_MIRROR}" sh /usr/local/sbin/apt-mirror && \
|
||||
apt-get update && \
|
||||
apt-get install -y --no-install-recommends \
|
||||
ca-certificates curl wget espeak-ng libgomp1 \
|
||||
ffmpeg libopenblas0 libopenblas-dev libopus0 sox && \
|
||||
ffmpeg libopenblas0 libopenblas-dev libopus0 sox \
|
||||
hwdata && \
|
||||
apt-get clean && \
|
||||
rm -rf /var/lib/apt/lists/*
|
||||
|
||||
@@ -171,6 +175,17 @@ RUN if [ "${BUILD_TYPE}" = "hipblas" ]; then \
|
||||
ln -s /opt/rocm-**/lib/llvm/lib/libomp.so /usr/lib/libomp.so \
|
||||
; fi
|
||||
|
||||
# ROCm's bundled libdrm_amdgpu is built with a hardcoded fallback lookup path
|
||||
# for the ASIC ID table (/opt/amdgpu/share/libdrm/amdgpu.ids), which only exists
|
||||
# if AMD's full amdgpu graphics/DKMS stack is installed. This compute-only image
|
||||
# doesn't have it, so hipblas/rocBLAS log "No such file or directory" on every
|
||||
# model load and can fail to identify the GPU. Point it at the equivalent file
|
||||
# Ubuntu's libdrm-common package already ships.
|
||||
RUN if [ "${BUILD_TYPE}" = "hipblas" ] && [ -f /usr/share/libdrm/amdgpu.ids ] && [ ! -e /opt/amdgpu/share/libdrm/amdgpu.ids ]; then \
|
||||
mkdir -p /opt/amdgpu/share/libdrm && \
|
||||
ln -s /usr/share/libdrm/amdgpu.ids /opt/amdgpu/share/libdrm/amdgpu.ids \
|
||||
; fi
|
||||
|
||||
RUN expr "${BUILD_TYPE}" = intel && echo "intel" > /run/localai/capability || echo "not intel"
|
||||
|
||||
# Cuda
|
||||
@@ -378,7 +393,12 @@ RUN go install github.com/mikefarah/yq/v4@latest
|
||||
# If you cannot find a more suitable place for an addition, this layer is a suitable place for it.
|
||||
FROM requirements-drivers
|
||||
|
||||
ENV HEALTHCHECK_ENDPOINT=http://localhost:8080/readyz
|
||||
# Optional override for the HEALTHCHECK target. Left empty so healthcheck.sh
|
||||
# derives the endpoint from the mode the container is actually running — the
|
||||
# same image runs `local-ai run` (HTTP on 8080) and `local-ai worker` (HTTP on
|
||||
# the gRPC base port minus one), and a hardcoded default marked every worker
|
||||
# permanently unhealthy (#10987). Set it to pin an explicit URL.
|
||||
ENV HEALTHCHECK_ENDPOINT=""
|
||||
|
||||
ARG CUDA_MAJOR_VERSION=12
|
||||
ENV NVIDIA_DRIVER_CAPABILITIES=compute,utility
|
||||
@@ -388,6 +408,7 @@ ENV NVIDIA_VISIBLE_DEVICES=all
|
||||
WORKDIR /
|
||||
|
||||
COPY ./entrypoint.sh .
|
||||
COPY ./scripts/build/healthcheck.sh .
|
||||
|
||||
# Copy the binary
|
||||
COPY --from=builder /build/local-ai ./
|
||||
@@ -398,9 +419,22 @@ RUN --mount=from=builder,src=/build/,dst=/mnt/build \
|
||||
# Make sure the models directory exists
|
||||
RUN mkdir -p /models /backends /data
|
||||
|
||||
# Define the health check command
|
||||
HEALTHCHECK --interval=1m --timeout=10m --retries=10 \
|
||||
CMD curl -f ${HEALTHCHECK_ENDPOINT} || exit 1
|
||||
# Define the health check command.
|
||||
#
|
||||
# --start-period is the knob for slow starts, not --timeout/--retries. Since
|
||||
# #10949 a frontend's startup preload materializes HuggingFace artifacts before
|
||||
# the HTTP server binds (31 GB observed on a live cluster), so a healthy replica
|
||||
# can legitimately fail probes for a long time. Failures inside the start period
|
||||
# leave the container `starting` instead of burning retries, and the period ends
|
||||
# early on the first success — so a generous value costs a fast-starting
|
||||
# container nothing. A process that actually died is handled by the restart
|
||||
# policy, not by health.
|
||||
#
|
||||
# --timeout is a per-probe deadline: 10m meant a wedged probe could hang for ten
|
||||
# minutes and stretch detection without bound. A localhost curl that has not
|
||||
# answered in 10s is itself the fault being detected.
|
||||
HEALTHCHECK --start-period=60m --interval=1m --timeout=10s --retries=3 \
|
||||
CMD /healthcheck.sh
|
||||
|
||||
VOLUME /models /backends /configuration /data
|
||||
EXPOSE 8080
|
||||
|
||||
103
Makefile
103
Makefile
@@ -1,5 +1,5 @@
|
||||
# Disable parallel execution for backend builds
|
||||
.NOTPARALLEL: backends/diffusers backends/llama-cpp backends/turboquant backends/outetts backends/piper backends/stablediffusion-ggml backends/whisper backends/crispasr backends/parakeet-cpp backends/faster-whisper backends/silero-vad backends/local-store backends/huggingface backends/rfdetr backends/rfdetr-cpp backends/insightface backends/speaker-recognition backends/kitten-tts backends/kokoro backends/chatterbox backends/llama-cpp-darwin backends/neutts build-darwin-python-backend build-darwin-go-backend backends/mlx backends/diffuser-darwin backends/mlx-vlm backends/mlx-audio backends/mlx-distributed backends/stablediffusion-ggml-darwin backends/vllm backends/vllm-omni backends/sglang backends/moonshine backends/pocket-tts backends/qwen-tts backends/faster-qwen3-tts backends/qwen-asr backends/nemo backends/voxcpm backends/whisperx backends/ace-step backends/acestep-cpp backends/fish-speech backends/voxtral backends/opus backends/trl backends/llama-cpp-quantization backends/kokoros backends/sam3-cpp backends/qwen3-tts-cpp backends/omnivoice-cpp backends/vibevoice-cpp backends/localvqe backends/tinygrad backends/sherpa-onnx backends/ds4 backends/ds4-darwin backends/liquid-audio backends/supertonic backends/depth-anything-cpp backends/privacy-filter backends/privacy-filter-darwin backends/llama-cpp-localai-paged
|
||||
.NOTPARALLEL: backends/diffusers backends/llama-cpp backends/turboquant backends/bonsai backends/outetts backends/piper backends/stablediffusion-ggml backends/whisper backends/crispasr backends/parakeet-cpp backends/moss-transcribe-cpp backends/faster-whisper backends/silero-vad backends/local-store backends/cloud-proxy backends/huggingface backends/rfdetr backends/rfdetr-cpp backends/insightface backends/speaker-recognition backends/kitten-tts backends/kokoro backends/chatterbox backends/llama-cpp-darwin backends/neutts build-darwin-python-backend build-darwin-go-backend backends/mlx backends/diffuser-darwin backends/mlx-vlm backends/mlx-audio backends/mlx-distributed backends/stablediffusion-ggml-darwin backends/vllm backends/vllm-omni backends/longcat-video backends/sglang backends/moonshine backends/pocket-tts backends/qwen-tts backends/faster-qwen3-tts backends/qwen-asr backends/nemo backends/voxcpm backends/whisperx backends/ace-step backends/acestep-cpp backends/fish-speech backends/voxtral backends/opus backends/trl backends/llama-cpp-quantization backends/kokoros backends/sam3-cpp backends/qwen3-tts-cpp backends/moss-tts-cpp backends/omnivoice-cpp backends/vibevoice-cpp backends/localvqe backends/tinygrad backends/sherpa-onnx backends/ds4 backends/ds4-darwin backends/liquid-audio backends/supertonic backends/depth-anything-cpp backends/privacy-filter backends/privacy-filter-darwin
|
||||
|
||||
GOCMD=go
|
||||
GOTEST=$(GOCMD) test
|
||||
@@ -103,7 +103,7 @@ COVERAGE_E2E_LABELS?=!real-models
|
||||
COVERAGE_EXCLUDE_RE?=grpc/proto/.*[.]pb[.]go
|
||||
|
||||
|
||||
.PHONY: all test test-coverage test-coverage-baseline test-coverage-check test-backend-cpp test-ui test-ui-coverage-baseline test-ui-coverage-check install-hooks build vendor lint lint-all
|
||||
.PHONY: all test test-coverage test-coverage-baseline test-coverage-check test-backend-cpp test-build-scripts test-ui test-ui-coverage-baseline test-ui-coverage-check install-hooks build vendor lint lint-all
|
||||
|
||||
all: help
|
||||
|
||||
@@ -208,6 +208,20 @@ test: prepare-test
|
||||
test-backend-cpp:
|
||||
bash backend/cpp/run-unit-tests.sh
|
||||
|
||||
## Runs the shell-level regression tests for the image packaging scripts
|
||||
## (scripts/build/*_test.sh). These guard invariants that only ever break
|
||||
## inside a container build - a missing transitive dep, a partial cuDNN
|
||||
## family - and that no Go test can observe. Needs only bash + gcc + ldd.
|
||||
test-build-scripts:
|
||||
@set -e; for t in scripts/build/*_test.sh; do echo "== $$t"; bash "$$t"; done
|
||||
|
||||
## Runs the unit tests for the CI helper scripts under scripts/lib/. Currently
|
||||
## the backend matrix path filter, whose failure mode is invisible in CI: it
|
||||
## emits an empty matrix, every job goes green, and the change ships to no
|
||||
## image at all (see PR #10946). Plain `node --test`, no dependencies.
|
||||
test-ci-scripts:
|
||||
@set -e; for t in scripts/lib/*_test.mjs; do echo "== $$t"; node --test "$$t"; done
|
||||
|
||||
## Runs the core suite ($(TEST_PATHS)) with statement-coverage instrumentation
|
||||
## and writes a merged profile to $(COVERAGE_PROFILE). Deliberately omits
|
||||
## --fail-fast so a single failure doesn't truncate the coverage number, and
|
||||
@@ -405,6 +419,23 @@ test-realtime: build-mock-backend
|
||||
@echo 'Running realtime e2e tests (mock backend)'
|
||||
$(GOCMD) run github.com/onsi/ginkgo/v2/ginkgo --label-filter="Realtime && !real-models" --flake-attempts $(TEST_FLAKES) -v -r ./tests/e2e
|
||||
|
||||
# Verify the realtime state-machine implementations conform to their formal
|
||||
# designs (Go transition/rapid tests under -race + FizzBee model check of the
|
||||
# authoritative specs). See docs/design/realtime-state-machines.md (Part 6) and
|
||||
# docs/design/specs/README.md.
|
||||
test-realtime-conformance:
|
||||
GOCMD=$(GOCMD) ./scripts/realtime-conformance.sh
|
||||
|
||||
# Verify the shared model-loader shutdown behavior independently of any API
|
||||
# modality (focused loader/gRPC/distributed/worker tests under -race + FizzBee).
|
||||
test-model-lifecycle-conformance:
|
||||
GOCMD=$(GOCMD) ./scripts/model-lifecycle-conformance.sh
|
||||
|
||||
# Install the pinned, checksum-verified FizzBee model checker (into .tools/,
|
||||
# gitignored) used by the conformance targets. Idempotent; no-op if present.
|
||||
install-fizzbee:
|
||||
./scripts/install-fizzbee.sh
|
||||
|
||||
# Container-based real-model realtime testing. Build env vars / pipeline
|
||||
# definition kept here so test-realtime-models-docker can drive a fully wired
|
||||
# pipeline (VAD + STT + LLM + TTS) from inside a containerised runner.
|
||||
@@ -553,6 +584,7 @@ prepare-test-extra: protogen-python
|
||||
$(MAKE) -C backend/python/chatterbox
|
||||
$(MAKE) -C backend/python/vllm
|
||||
$(MAKE) -C backend/python/vllm-omni
|
||||
$(MAKE) -C backend/python/longcat-video
|
||||
$(MAKE) -C backend/python/sglang
|
||||
$(MAKE) -C backend/python/vibevoice
|
||||
$(MAKE) -C backend/python/liquid-audio
|
||||
@@ -582,6 +614,7 @@ test-extra: prepare-test-extra
|
||||
$(MAKE) -C backend/python/chatterbox test
|
||||
$(MAKE) -C backend/python/vllm test
|
||||
$(MAKE) -C backend/python/vllm-omni test
|
||||
$(MAKE) -C backend/python/longcat-video test
|
||||
$(MAKE) -C backend/python/vibevoice test
|
||||
$(MAKE) -C backend/python/liquid-audio test
|
||||
$(MAKE) -C backend/python/moonshine test
|
||||
@@ -633,6 +666,9 @@ test-extra: prepare-test-extra
|
||||
## suite against it.
|
||||
##
|
||||
BACKEND_TEST_MODEL_URL?=https://huggingface.co/Qwen/Qwen3-0.6B-GGUF/resolve/main/Qwen3-0.6B-Q8_0.gguf
|
||||
## Suite timeout for `go test`. Wrappers whose model download alone can eat
|
||||
## most of the default (multi-GB models on a slow HF CDN day) override this.
|
||||
BACKEND_TEST_TIMEOUT?=30m
|
||||
|
||||
## Generic target — runs the suite against whatever BACKEND_IMAGE points at.
|
||||
## Depends on protogen-go so pkg/grpc/proto is generated before `go test`.
|
||||
@@ -660,7 +696,7 @@ test-extra-backend: protogen-go
|
||||
BACKEND_TEST_FACE_IMAGE_3_URL="$$BACKEND_TEST_FACE_IMAGE_3_URL" \
|
||||
BACKEND_TEST_FACE_IMAGE_3_FILE="$$BACKEND_TEST_FACE_IMAGE_3_FILE" \
|
||||
BACKEND_TEST_VERIFY_DISTANCE_CEILING="$$BACKEND_TEST_VERIFY_DISTANCE_CEILING" \
|
||||
go test -v -timeout 30m ./tests/e2e-backends/...
|
||||
go test -v -timeout $(BACKEND_TEST_TIMEOUT) ./tests/e2e-backends/...
|
||||
|
||||
## Convenience wrappers: build the image, then exercise it.
|
||||
test-extra-backend-llama-cpp: docker-build-llama-cpp
|
||||
@@ -671,15 +707,6 @@ test-extra-backend-llama-cpp: docker-build-llama-cpp
|
||||
test-extra-backend-ik-llama-cpp: docker-build-ik-llama-cpp
|
||||
BACKEND_IMAGE=local-ai-backend:ik-llama-cpp $(MAKE) test-extra-backend
|
||||
|
||||
## llama-cpp-localai-paged: the LocalAI paged-attention llama.cpp variant. Same
|
||||
## GGUF surface as stock llama-cpp (the paged engine is runtime-gated by the
|
||||
## LLAMA_KV_PAGED env the grpc-server option hooks set), so the standard
|
||||
## llama-cpp capability set is what we exercise here.
|
||||
test-extra-backend-llama-cpp-localai-paged: docker-build-llama-cpp-localai-paged
|
||||
BACKEND_IMAGE=local-ai-backend:llama-cpp-localai-paged \
|
||||
BACKEND_TEST_CAPS=health,load,predict,stream,logprobs,logit_bias \
|
||||
$(MAKE) test-extra-backend
|
||||
|
||||
## turboquant: exercises the llama.cpp-fork backend with the fork's
|
||||
## *TurboQuant-specific* KV-cache types (turbo3 for both K and V). turbo3
|
||||
## is what makes this backend distinct from stock llama-cpp — picking q8_0
|
||||
@@ -692,6 +719,16 @@ test-extra-backend-turboquant: docker-build-turboquant
|
||||
BACKEND_TEST_CACHE_TYPE_V=turbo3 \
|
||||
$(MAKE) test-extra-backend
|
||||
|
||||
## bonsai: exercises the llama.cpp-fork backend with a real Q1_0 (1-bit) model —
|
||||
## the PrismML Bonsai-8B GGUF, whose weight quant is *only* decodable by the fork's
|
||||
## Q1_0 kernels. Loading it is what makes this backend distinct from stock llama-cpp;
|
||||
## a standard-quant model would only test the upstream code path the llama-cpp backend
|
||||
## already covers.
|
||||
test-extra-backend-bonsai: docker-build-bonsai
|
||||
BACKEND_IMAGE=local-ai-backend:bonsai \
|
||||
BACKEND_TEST_MODEL_URL=https://huggingface.co/prism-ml/Bonsai-8B-gguf/resolve/main/Bonsai-8B-Q1_0.gguf \
|
||||
$(MAKE) test-extra-backend
|
||||
|
||||
## Audio transcription wrapper for the llama-cpp backend.
|
||||
## Drives the new AudioTranscription / AudioTranscriptionStream RPCs against
|
||||
## ggml-org/Qwen3-ASR-0.6B-GGUF (a small ASR model that requires its mmproj
|
||||
@@ -1009,6 +1046,7 @@ test-extra-backend-vibevoice-cpp-tts: docker-build-vibevoice-cpp
|
||||
## post-image disk budget.
|
||||
test-extra-backend-vibevoice-cpp-transcription: docker-build-vibevoice-cpp
|
||||
BACKEND_IMAGE=local-ai-backend:vibevoice-cpp \
|
||||
BACKEND_TEST_TIMEOUT=120m \
|
||||
BACKEND_TEST_MODEL_URL='https://huggingface.co/mudler/vibevoice.cpp-models/resolve/main/vibevoice-asr-q4_k.gguf#vibevoice-asr-q4_k.gguf' \
|
||||
BACKEND_TEST_EXTRA_FILES='https://huggingface.co/mudler/vibevoice.cpp-models/resolve/main/tokenizer.gguf#tokenizer.gguf' \
|
||||
BACKEND_TEST_AUDIO_URL=https://github.com/ggml-org/whisper.cpp/raw/master/samples/jfk.wav \
|
||||
@@ -1036,7 +1074,19 @@ test-extra-backend-whisper-transcription: docker-build-whisper
|
||||
## is reachable.
|
||||
test-extra-backend-parakeet-cpp-transcription: docker-build-parakeet-cpp
|
||||
BACKEND_IMAGE=local-ai-backend:parakeet-cpp \
|
||||
BACKEND_TEST_MODEL_URL=https://huggingface.co/mudler/parakeet-cpp-gguf/resolve/main/tdt_ctc-110m-f16.gguf \
|
||||
BACKEND_TEST_MODEL_URL=https://huggingface.co/mudler/parakeet-cpp-gguf/resolve/main/realtime_eou_120m-v1-f16.gguf \
|
||||
BACKEND_TEST_AUDIO_URL=https://github.com/ggml-org/whisper.cpp/raw/master/samples/jfk.wav \
|
||||
BACKEND_TEST_CAPS=health,load,transcription \
|
||||
$(MAKE) test-extra-backend
|
||||
|
||||
## Audio transcription wrapper for the moss-transcribe-cpp (moss-transcribe.cpp
|
||||
## ggml port) backend. Mirrors test-extra-backend-parakeet-cpp-transcription:
|
||||
## drives the AudioTranscription RPC against a published MOSS GGUF using the JFK
|
||||
## 11s clip from whisper.cpp's CI samples. Not part of the default test suite -
|
||||
## run explicitly once the pinned model URL is reachable.
|
||||
test-extra-backend-moss-transcribe-cpp-transcription: docker-build-moss-transcribe-cpp
|
||||
BACKEND_IMAGE=local-ai-backend:moss-transcribe-cpp \
|
||||
BACKEND_TEST_MODEL_URL=https://huggingface.co/mudler/moss-transcribe.cpp-gguf/resolve/main/moss-transcribe-q5_k.gguf \
|
||||
BACKEND_TEST_AUDIO_URL=https://github.com/ggml-org/whisper.cpp/raw/master/samples/jfk.wav \
|
||||
BACKEND_TEST_CAPS=health,load,transcription \
|
||||
$(MAKE) test-extra-backend
|
||||
@@ -1190,10 +1240,10 @@ BACKEND_IK_LLAMA_CPP = ik-llama-cpp|ik-llama-cpp|.|false|false
|
||||
# turboquant is a llama.cpp fork with TurboQuant KV-cache quantization.
|
||||
# Reuses backend/cpp/llama-cpp grpc-server sources via a thin wrapper Makefile.
|
||||
BACKEND_TURBOQUANT = turboquant|turboquant|.|false|false
|
||||
# llama-cpp-localai-paged = stock llama.cpp grpc-server + the LocalAI paged-attention
|
||||
# patch series (vendored in this wrapper backend). Reuses backend/cpp/llama-cpp sources via a thin
|
||||
# wrapper Makefile (same upstream pin as stock llama-cpp; no fork, no patch-grpc-server).
|
||||
BACKEND_LLAMA_CPP_LOCALAI_PAGED = llama-cpp-localai-paged|llama-cpp-localai-paged|.|false|false
|
||||
# bonsai is a llama.cpp fork (PrismML) adding the Q1_0 (1-bit) and Q2_0 (ternary)
|
||||
# weight-quant kernels the Bonsai / Ternary-Bonsai models ship in. Reuses
|
||||
# backend/cpp/llama-cpp grpc-server sources via a thin wrapper Makefile.
|
||||
BACKEND_BONSAI = bonsai|bonsai|.|false|false
|
||||
# ds4 is antirez/ds4, a DeepSeek V4 Flash-specific inference engine.
|
||||
# Single-model; hardware-only validation lives at tests/e2e-backends/
|
||||
# (BACKEND_BINARY mode); see docs/superpowers/plans/2026-05-11-ds4-backend.md.
|
||||
@@ -1213,10 +1263,12 @@ BACKEND_STABLEDIFFUSION_GGML = stablediffusion-ggml|golang|.|--progress=plain|tr
|
||||
BACKEND_WHISPER = whisper|golang|.|false|true
|
||||
BACKEND_CRISPASR = crispasr|golang|.|false|true
|
||||
BACKEND_PARAKEET_CPP = parakeet-cpp|golang|.|false|true
|
||||
BACKEND_MOSS_TRANSCRIBE_CPP = moss-transcribe-cpp|golang|.|false|true
|
||||
BACKEND_DEPTH_ANYTHING_CPP = depth-anything-cpp|golang|.|false|true
|
||||
BACKEND_VOXTRAL = voxtral|golang|.|false|true
|
||||
BACKEND_ACESTEP_CPP = acestep-cpp|golang|.|false|true
|
||||
BACKEND_QWEN3_TTS_CPP = qwen3-tts-cpp|golang|.|false|true
|
||||
BACKEND_MOSS_TTS_CPP = moss-tts-cpp|golang|.|false|true
|
||||
BACKEND_OMNIVOICE_CPP = omnivoice-cpp|golang|.|false|true
|
||||
BACKEND_VIBEVOICE_CPP = vibevoice-cpp|golang|.|false|true
|
||||
BACKEND_LOCALVQE = localvqe|golang|.|false|true
|
||||
@@ -1238,6 +1290,7 @@ BACKEND_NEUTTS = neutts|python|.|false|true
|
||||
BACKEND_KOKORO = kokoro|python|.|false|true
|
||||
BACKEND_VLLM = vllm|python|.|false|true
|
||||
BACKEND_VLLM_OMNI = vllm-omni|python|.|false|true
|
||||
BACKEND_LONGCAT_VIDEO = longcat-video|python|.|--progress=plain|true
|
||||
BACKEND_SGLANG = sglang|python|.|false|true
|
||||
BACKEND_DIFFUSERS = diffusers|python|.|--progress=plain|true
|
||||
BACKEND_CHATTERBOX = chatterbox|python|.|false|true
|
||||
@@ -1295,7 +1348,7 @@ endef
|
||||
$(eval $(call generate-docker-build-target,$(BACKEND_LLAMA_CPP)))
|
||||
$(eval $(call generate-docker-build-target,$(BACKEND_IK_LLAMA_CPP)))
|
||||
$(eval $(call generate-docker-build-target,$(BACKEND_TURBOQUANT)))
|
||||
$(eval $(call generate-docker-build-target,$(BACKEND_LLAMA_CPP_LOCALAI_PAGED)))
|
||||
$(eval $(call generate-docker-build-target,$(BACKEND_BONSAI)))
|
||||
$(eval $(call generate-docker-build-target,$(BACKEND_DS4)))
|
||||
$(eval $(call generate-docker-build-target,$(BACKEND_PRIVACY_FILTER)))
|
||||
$(eval $(call generate-docker-build-target,$(BACKEND_PIPER)))
|
||||
@@ -1307,6 +1360,7 @@ $(eval $(call generate-docker-build-target,$(BACKEND_STABLEDIFFUSION_GGML)))
|
||||
$(eval $(call generate-docker-build-target,$(BACKEND_WHISPER)))
|
||||
$(eval $(call generate-docker-build-target,$(BACKEND_CRISPASR)))
|
||||
$(eval $(call generate-docker-build-target,$(BACKEND_PARAKEET_CPP)))
|
||||
$(eval $(call generate-docker-build-target,$(BACKEND_MOSS_TRANSCRIBE_CPP)))
|
||||
$(eval $(call generate-docker-build-target,$(BACKEND_DEPTH_ANYTHING_CPP)))
|
||||
$(eval $(call generate-docker-build-target,$(BACKEND_VOXTRAL)))
|
||||
$(eval $(call generate-docker-build-target,$(BACKEND_OPUS)))
|
||||
@@ -1323,6 +1377,7 @@ $(eval $(call generate-docker-build-target,$(BACKEND_NEUTTS)))
|
||||
$(eval $(call generate-docker-build-target,$(BACKEND_KOKORO)))
|
||||
$(eval $(call generate-docker-build-target,$(BACKEND_VLLM)))
|
||||
$(eval $(call generate-docker-build-target,$(BACKEND_VLLM_OMNI)))
|
||||
$(eval $(call generate-docker-build-target,$(BACKEND_LONGCAT_VIDEO)))
|
||||
$(eval $(call generate-docker-build-target,$(BACKEND_SGLANG)))
|
||||
$(eval $(call generate-docker-build-target,$(BACKEND_DIFFUSERS)))
|
||||
$(eval $(call generate-docker-build-target,$(BACKEND_CHATTERBOX)))
|
||||
@@ -1340,6 +1395,7 @@ $(eval $(call generate-docker-build-target,$(BACKEND_WHISPERX)))
|
||||
$(eval $(call generate-docker-build-target,$(BACKEND_ACE_STEP)))
|
||||
$(eval $(call generate-docker-build-target,$(BACKEND_ACESTEP_CPP)))
|
||||
$(eval $(call generate-docker-build-target,$(BACKEND_QWEN3_TTS_CPP)))
|
||||
$(eval $(call generate-docker-build-target,$(BACKEND_MOSS_TTS_CPP)))
|
||||
$(eval $(call generate-docker-build-target,$(BACKEND_OMNIVOICE_CPP)))
|
||||
$(eval $(call generate-docker-build-target,$(BACKEND_VIBEVOICE_CPP)))
|
||||
$(eval $(call generate-docker-build-target,$(BACKEND_LOCALVQE)))
|
||||
@@ -1359,7 +1415,7 @@ $(eval $(call generate-docker-build-target,$(BACKEND_SUPERTONIC)))
|
||||
docker-save-%: backend-images
|
||||
docker save local-ai-backend:$* -o backend-images/$*.tar
|
||||
|
||||
docker-build-backends: docker-build-llama-cpp docker-build-ik-llama-cpp docker-build-turboquant docker-build-llama-cpp-localai-paged docker-build-ds4 docker-build-rerankers docker-build-vllm docker-build-vllm-omni docker-build-sglang docker-build-transformers docker-build-outetts docker-build-diffusers docker-build-kokoro docker-build-faster-whisper docker-build-crispasr docker-build-coqui docker-build-chatterbox docker-build-vibevoice docker-build-liquid-audio docker-build-moonshine docker-build-pocket-tts docker-build-qwen-tts docker-build-fish-speech docker-build-faster-qwen3-tts docker-build-qwen-asr docker-build-nemo docker-build-voxcpm docker-build-whisperx docker-build-ace-step docker-build-acestep-cpp docker-build-voxtral docker-build-mlx-distributed docker-build-trl docker-build-llama-cpp-quantization docker-build-tinygrad docker-build-kokoros docker-build-sam3-cpp docker-build-rfdetr-cpp docker-build-qwen3-tts-cpp docker-build-omnivoice-cpp docker-build-vibevoice-cpp docker-build-localvqe docker-build-insightface docker-build-speaker-recognition docker-build-sherpa-onnx docker-build-cloud-proxy docker-build-supertonic docker-build-depth-anything-cpp docker-build-privacy-filter
|
||||
docker-build-backends: docker-build-llama-cpp docker-build-ik-llama-cpp docker-build-turboquant docker-build-bonsai docker-build-ds4 docker-build-rerankers docker-build-vllm docker-build-vllm-omni docker-build-longcat-video docker-build-sglang docker-build-transformers docker-build-outetts docker-build-diffusers docker-build-kokoro docker-build-faster-whisper docker-build-crispasr docker-build-coqui docker-build-chatterbox docker-build-vibevoice docker-build-liquid-audio docker-build-moonshine docker-build-pocket-tts docker-build-qwen-tts docker-build-fish-speech docker-build-faster-qwen3-tts docker-build-qwen-asr docker-build-nemo docker-build-voxcpm docker-build-whisperx docker-build-ace-step docker-build-acestep-cpp docker-build-voxtral docker-build-mlx-distributed docker-build-trl docker-build-llama-cpp-quantization docker-build-tinygrad docker-build-kokoros docker-build-sam3-cpp docker-build-rfdetr-cpp docker-build-qwen3-tts-cpp docker-build-moss-tts-cpp docker-build-omnivoice-cpp docker-build-vibevoice-cpp docker-build-localvqe docker-build-insightface docker-build-speaker-recognition docker-build-sherpa-onnx docker-build-cloud-proxy docker-build-supertonic docker-build-depth-anything-cpp docker-build-moss-transcribe-cpp docker-build-privacy-filter
|
||||
|
||||
########################################################
|
||||
### Mock Backend for E2E Tests
|
||||
@@ -1484,8 +1540,13 @@ build-launcher-darwin:
|
||||
mv cmd/launcher/LocalAI.app dist/LocalAI.app
|
||||
bash contrib/macos/sign-and-notarize.sh sign dist/LocalAI.app
|
||||
|
||||
# Wrap the (signed) app into a drag-to-Applications DMG via hdiutil, then sign the DMG.
|
||||
# Notarize + staple the .app itself, then wrap it into a drag-to-Applications
|
||||
# DMG via hdiutil and sign the DMG. The app is stapled BEFORE packaging so the
|
||||
# bundle carries its own ticket and verifies offline (a dmg-only staple leaves
|
||||
# the app relying on an online Gatekeeper check, which fails offline / once the
|
||||
# app is copied out of the dmg). No-op without notary secrets.
|
||||
dmg-launcher-darwin: build-launcher-darwin
|
||||
bash contrib/macos/sign-and-notarize.sh notarize-app dist/LocalAI.app
|
||||
rm -rf dist/dmg dist/LocalAI.dmg
|
||||
mkdir -p dist/dmg
|
||||
cp -R dist/LocalAI.app dist/dmg/LocalAI.app
|
||||
@@ -1497,7 +1558,7 @@ dmg-launcher-darwin: build-launcher-darwin
|
||||
notarize-launcher-darwin: dmg-launcher-darwin
|
||||
bash contrib/macos/sign-and-notarize.sh notarize dist/LocalAI.dmg
|
||||
|
||||
# Single entrypoint for CI: build -> sign app -> dmg -> sign dmg -> notarize -> staple.
|
||||
# Single entrypoint for CI: build -> sign app -> notarize+staple app -> dmg -> sign dmg -> notarize+staple dmg.
|
||||
release-launcher-darwin: notarize-launcher-darwin
|
||||
@echo "dist/LocalAI.dmg is ready"
|
||||
|
||||
|
||||
13
README.md
13
README.md
@@ -177,7 +177,7 @@ For more details, see the [Getting Started guide](https://localai.io/basics/gett
|
||||
|
||||
## Latest News
|
||||
|
||||
- **June 2026**: New native biometric backends from the LocalAI team: [voice-detect.cpp](https://github.com/mudler/voice-detect.cpp) for speaker recognition and voice analysis (ECAPA-TDNN, WeSpeaker, ERes2Net, CAM++, wav2vec2 age/gender/emotion) and [face-detect.cpp](https://github.com/mudler/face-detect.cpp) for face detection, recognition, demographics and anti-spoofing (SCRFD/ArcFace, YuNet/SFace). Both are from-scratch C++/ggml engines with no Python or onnxruntime at inference, self-contained GGUF weights, bit-exact parity with the reference, and GPU cuDNN parity, replacing the heavier Python `insightface` and `speaker-recognition` backends ([PR #10441](https://github.com/mudler/LocalAI/pull/10441)).
|
||||
- **June 2026**: New native biometric backends from the LocalAI team: [voice-detect.cpp](https://github.com/localai-org/voice-detect.cpp) for speaker recognition and voice analysis (ECAPA-TDNN, WeSpeaker, ERes2Net, CAM++, wav2vec2 age/gender/emotion) and [face-detect.cpp](https://github.com/mudler/face-detect.cpp) for face detection, recognition, demographics and anti-spoofing (SCRFD/ArcFace, YuNet/SFace). Both are from-scratch C++/ggml engines with no Python or onnxruntime at inference, self-contained GGUF weights, bit-exact parity with the reference, and GPU cuDNN parity, replacing the heavier Python `insightface` and `speaker-recognition` backends ([PR #10441](https://github.com/mudler/LocalAI/pull/10441)).
|
||||
- **June 2026**: New [realtime voice assistant demo](https://github.com/localai-org/localai-realtime-demo) (a tiny Go client for the Realtime API with a full talk-back voice loop and tool calling), plus [streaming of the realtime LLM / TTS / transcription pipeline stages](https://github.com/mudler/LocalAI/pull/10176) and [configurable WebRTC ICE candidates](https://github.com/mudler/LocalAI/pull/10231).
|
||||
- **June 2026**: Big speech push: the [parakeet.cpp](https://github.com/mudler/parakeet.cpp) ASR engine gains [NeMo-faithful segment timestamps](https://github.com/mudler/LocalAI/pull/10207), a [multilingual streaming Nemotron-3.5 model](https://github.com/mudler/LocalAI/pull/10199), [dynamic batching for concurrent transcription](https://github.com/mudler/LocalAI/pull/10112) and [CUDA graphs](https://github.com/mudler/LocalAI/pull/10273); the new [CrispASR backend](https://github.com/mudler/LocalAI/pull/10099) adds multi-architecture ASR + TTS, and [60 Piper TTS voices across 42 languages](https://github.com/mudler/LocalAI/pull/10296) land in the gallery (plus [per-request TTS instructions and params](https://github.com/mudler/LocalAI/pull/10172)).
|
||||
- **June 2026**: New backends and models: [locate-anything.cpp](https://github.com/mudler/LocalAI/pull/10264) for open-vocabulary object detection via ggml, [Ideogram4 image generation](https://github.com/mudler/LocalAI/pull/10201) in stablediffusion-ggml, [llama.cpp video input](https://github.com/mudler/LocalAI/pull/10216), and the [Gemma 4 QAT family with MTP speculative-decoding pairs](https://github.com/mudler/LocalAI/pull/10215). Plus an [interactive CLI chat mode](https://github.com/mudler/LocalAI/pull/10226) and [RAG source citations in agent responses](https://github.com/mudler/LocalAI/pull/10228).
|
||||
@@ -232,12 +232,17 @@ Most backends wrap a best-in-class upstream engine. A handful of them are native
|
||||
| Backend | What it does |
|
||||
|---------|-------------|
|
||||
| [parakeet.cpp](https://github.com/mudler/parakeet.cpp) | C++/GGML port of NVIDIA NeMo Parakeet ASR (tdt/ctc/rnnt/hybrid), with cache-aware streaming transcription |
|
||||
| [ced.cpp](https://github.com/mudler/ced.cpp) | C++/GGML port of the CED audio-tagging models: sound-event classification (527-class AudioSet) over REST and the realtime API for live recognition |
|
||||
| [voxtral.c](https://github.com/mudler/voxtral.c) | Voxtral Realtime 4B speech-to-text in pure C |
|
||||
| [moss-transcribe.cpp](https://github.com/localai-org/moss-transcribe.cpp) | C++/GGML port of OpenMOSS MOSS-Transcribe-Diarize: joint long-form transcription, speaker diarization and timestamping in a single pass |
|
||||
| [moss-tts.cpp](https://github.com/mudler/moss-tts.cpp) | C++/GGML port of the OpenMOSS MOSS-TTS family: text-to-speech (MOSS-TTS-Local v1.5, 48 kHz stereo) with reference-audio voice cloning, through the MOSS-Audio-Tokenizer neural codec |
|
||||
| [ced.cpp](https://github.com/localai-org/ced.cpp) | C++/GGML port of the CED audio-tagging models: sound-event classification (527-class AudioSet) over REST and the realtime API for live recognition |
|
||||
| [voice-detect.cpp](https://github.com/localai-org/voice-detect.cpp) | Speaker recognition and voice analysis (ECAPA-TDNN, WeSpeaker, ERes2Net, CAM++, wav2vec2 age/gender/emotion), replacing the Python speaker-recognition backend |
|
||||
| [voxtral-tts.c](https://github.com/mudler/voxtral-tts.c) | Voxtral Realtime 4B speech-to-text in pure C |
|
||||
| [vibevoice.cpp](https://github.com/mudler/vibevoice.cpp) | Native port of Microsoft VibeVoice for TTS (voice cloning) and long-form ASR with speaker diarization |
|
||||
| [rf-detr.cpp](https://github.com/mudler/rf-detr.cpp) | Native RF-DETR object detection and instance segmentation |
|
||||
| [rf-detr.cpp](https://github.com/localai-org/rf-detr.cpp) | Native RF-DETR object detection and instance segmentation |
|
||||
| [locate-anything.cpp](https://github.com/mudler/locate-anything.cpp) | Open-vocabulary object detection and visual grounding (LocateAnything-3B) |
|
||||
| [depth-anything.cpp](https://github.com/mudler/depth-anything.cpp) | Depth Anything 3 monocular metric depth + camera pose estimation |
|
||||
| [face-detect.cpp](https://github.com/mudler/face-detect.cpp) | Face detection, recognition, demographics and anti-spoofing (SCRFD/ArcFace, YuNet/SFace), replacing the Python insightface backend |
|
||||
| [free-splatter.cpp](https://github.com/localai-org/free-splatter.cpp) | Pose-free 3D reconstruction (FreeSplatter): turns a handful of plain photos into 3D Gaussians, no camera poses or GPU required |
|
||||
| [privacy-filter.cpp](https://github.com/localai-org/privacy-filter.cpp) | Standalone GGML PII/NER token-classification engine powering LocalAI's PII redaction tier |
|
||||
| [LocalVQE](https://github.com/localai-org/LocalVQE) | Joint acoustic echo cancellation, noise suppression, and dereverberation |
|
||||
| [local-store](https://github.com/mudler/LocalAI) | Local-first vector database for embeddings (shipped in-tree) |
|
||||
|
||||
@@ -7,7 +7,7 @@ ARG BUILDER_BASE_IMAGE=${BASE_IMAGE}
|
||||
# BUILDER_TARGET selects which builder stage the final scratch image copies
|
||||
# package output from. Declared at global scope (before any FROM) so it's
|
||||
# usable in `FROM ${BUILDER_TARGET}` below. Default keeps local
|
||||
# `make backends/llama-cpp-localai-paged` on the from-source path.
|
||||
# `make backends/bonsai` on the from-source path.
|
||||
ARG BUILDER_TARGET=builder-fromsource
|
||||
ARG APT_MIRROR=""
|
||||
ARG APT_PORTS_MIRROR=""
|
||||
@@ -18,7 +18,7 @@ ARG APT_PORTS_MIRROR=""
|
||||
# Runs .docker/install-base-deps.sh (apt deps + cmake + protoc + gRPC +
|
||||
# conditional CUDA/ROCm/Vulkan), copies /opt/grpc to /usr/local, then
|
||||
# compiles the variant. Used when BUILDER_TARGET=builder-fromsource (the
|
||||
# default; local `make backends/llama-cpp-localai-paged`).
|
||||
# default; local `make backends/bonsai`).
|
||||
#
|
||||
# The install script is the same one that backend/Dockerfile.base-grpc-builder
|
||||
# runs, so the result is bit-equivalent to the prebuilt-base path
|
||||
@@ -84,22 +84,21 @@ RUN cp -a /opt/grpc/. /usr/local/
|
||||
COPY . /LocalAI
|
||||
|
||||
# BuildKit cache mount for ccache. See Dockerfile.llama-cpp (commit 9228e5b4)
|
||||
# for rationale. llama-cpp-localai-paged is the SAME upstream llama.cpp with
|
||||
# the LocalAI paged patch series applied; it reuses backend/cpp/llama-cpp
|
||||
# source via a thin wrapper Makefile, so MOST TUs are content-identical to the
|
||||
# stock llama-cpp build. Sharing a cache id with llama-cpp could give
|
||||
# cross-variant hits — but for now keep them separate (mirroring turboquant) so
|
||||
# a regression in one doesn't poison the other. Revisit sharing after measuring
|
||||
# the actual hit rate.
|
||||
# for rationale. bonsai is a llama.cpp fork that reuses
|
||||
# backend/cpp/llama-cpp source via a thin wrapper Makefile, so MOST TUs
|
||||
# are content-identical to the upstream llama-cpp build. Sharing a cache
|
||||
# id with llama-cpp could give cross-fork hits — but for now keep them
|
||||
# separate so a regression in one doesn't poison the other. Revisit
|
||||
# sharing after measuring the actual hit rate.
|
||||
#
|
||||
# The compile body is shared with builder-prebuilt via .docker/llama-cpp-localai-paged-compile.sh.
|
||||
RUN --mount=type=bind,source=.docker/llama-cpp-localai-paged-compile.sh,target=/usr/local/sbin/compile.sh \
|
||||
--mount=type=cache,target=/root/.ccache,id=llama-cpp-localai-paged-ccache-${TARGETARCH}-${BUILD_TYPE},sharing=locked \
|
||||
# The compile body is shared with builder-prebuilt via .docker/bonsai-compile.sh.
|
||||
RUN --mount=type=bind,source=.docker/bonsai-compile.sh,target=/usr/local/sbin/compile.sh \
|
||||
--mount=type=cache,target=/root/.ccache,id=bonsai-ccache-${TARGETARCH}-${BUILD_TYPE},sharing=locked \
|
||||
bash /usr/local/sbin/compile.sh
|
||||
|
||||
|
||||
# Copy libraries using a script to handle architecture differences
|
||||
RUN make -BC /LocalAI/backend/cpp/llama-cpp-localai-paged package
|
||||
RUN make -BC /LocalAI/backend/cpp/bonsai package
|
||||
|
||||
|
||||
# ============================================================================
|
||||
@@ -108,9 +107,7 @@ RUN make -BC /LocalAI/backend/cpp/llama-cpp-localai-paged package
|
||||
# That image already has gRPC at /opt/grpc + apt deps + CUDA/ROCm/Vulkan
|
||||
# pre-installed, so we just copy gRPC to /usr/local and compile. Used when
|
||||
# BUILDER_TARGET=builder-prebuilt (CI when the matrix entry sets
|
||||
# builder-base-image). llama-cpp-localai-paged reuses the SAME base-grpc-* tags
|
||||
# as the stock llama-cpp backend (same gRPC + same toolchain), so no new
|
||||
# base-images.yml variant is required.
|
||||
# builder-base-image).
|
||||
# ============================================================================
|
||||
FROM ${BUILDER_BASE_IMAGE} AS builder-prebuilt
|
||||
|
||||
@@ -121,9 +118,9 @@ ENV CUDA_DOCKER_ARCH=${CUDA_DOCKER_ARCH}
|
||||
ARG CMAKE_ARGS
|
||||
ENV CMAKE_ARGS=${CMAKE_ARGS}
|
||||
# AMDGPU_TARGETS must be forwarded into the env here too — backend/cpp/llama-cpp/Makefile
|
||||
# (which the llama-cpp-localai-paged Makefile reuses via a sibling build dir) errors out
|
||||
# when the var is empty on a hipblas build, and the prebuilt path is what CI exercises most
|
||||
# of the time. The builder-fromsource stage above already does this; mirror it here.
|
||||
# (which the bonsai Makefile reuses via a sibling build dir) errors out when the var
|
||||
# is empty on a hipblas build, and the prebuilt path is what CI exercises most of the
|
||||
# time. The builder-fromsource stage above already does this; mirror it here.
|
||||
ARG AMDGPU_TARGETS
|
||||
ENV AMDGPU_TARGETS=${AMDGPU_TARGETS}
|
||||
ARG TARGETARCH
|
||||
@@ -136,11 +133,11 @@ RUN cp -a /opt/grpc/. /usr/local/
|
||||
|
||||
COPY . /LocalAI
|
||||
|
||||
RUN --mount=type=bind,source=.docker/llama-cpp-localai-paged-compile.sh,target=/usr/local/sbin/compile.sh \
|
||||
--mount=type=cache,target=/root/.ccache,id=llama-cpp-localai-paged-ccache-${TARGETARCH}-${BUILD_TYPE},sharing=locked \
|
||||
RUN --mount=type=bind,source=.docker/bonsai-compile.sh,target=/usr/local/sbin/compile.sh \
|
||||
--mount=type=cache,target=/root/.ccache,id=bonsai-ccache-${TARGETARCH}-${BUILD_TYPE},sharing=locked \
|
||||
bash /usr/local/sbin/compile.sh
|
||||
|
||||
RUN make -BC /LocalAI/backend/cpp/llama-cpp-localai-paged package
|
||||
RUN make -BC /LocalAI/backend/cpp/bonsai package
|
||||
|
||||
|
||||
# ============================================================================
|
||||
@@ -160,4 +157,4 @@ FROM scratch
|
||||
|
||||
|
||||
# Copy all available binaries (the build process only creates the appropriate ones for the target architecture)
|
||||
COPY --from=builder /LocalAI/backend/cpp/llama-cpp-localai-paged/package/. ./
|
||||
COPY --from=builder /LocalAI/backend/cpp/bonsai/package/. ./
|
||||
@@ -224,7 +224,11 @@ ARG DEPS_REFRESH=initial
|
||||
|
||||
RUN cd /${BACKEND} && PORTABLE_PYTHON=true make
|
||||
|
||||
# Package GPU libraries into the backend's lib directory
|
||||
# Package GPU libraries into the backend's lib directory.
|
||||
#
|
||||
# Must stay after the venv is built above: package-gpu-libs.sh inspects
|
||||
# /${BACKEND}/venv to decide whether this backend already carries a complete
|
||||
# cuDNN from pip, and bundles one only when it does not (issue #10905).
|
||||
RUN mkdir -p /${BACKEND}/lib && \
|
||||
TARGET_LIB_DIR="/${BACKEND}/lib" BUILD_TYPE="${BUILD_TYPE}" CUDA_MAJOR_VERSION="${CUDA_MAJOR_VERSION}" \
|
||||
bash /package-gpu-libs.sh "/${BACKEND}/lib"
|
||||
|
||||
@@ -46,6 +46,7 @@ The backend system provides language-specific Dockerfiles that handle the build
|
||||
- **vllm**: High-performance LLM inference
|
||||
- **mlx**: Apple Silicon optimization
|
||||
- **diffusers**: Stable Diffusion models
|
||||
- **longcat-video**: CUDA text/image-to-video and speech-driven avatar generation
|
||||
- **Audio**: coqui, faster-whisper, kitten-tts
|
||||
- **Vision**: mlx-vlm, rfdetr
|
||||
- **Specialized**: rerankers, chatterbox, kokoro
|
||||
|
||||
@@ -18,6 +18,18 @@ service Backend {
|
||||
rpc GenerateVideo(GenerateVideoRequest) returns (Result) {}
|
||||
rpc AudioTranscription(TranscriptRequest) returns (TranscriptResult) {}
|
||||
rpc AudioTranscriptionStream(TranscriptRequest) returns (stream TranscriptStreamResponse) {}
|
||||
// AudioTranscriptionLive is the bidirectional live-microphone ASR RPC. The
|
||||
// first message MUST carry a Config; subsequent messages carry Audio frames
|
||||
// (mono float PCM at config.sample_rate, 16 kHz default). After a
|
||||
// successful open the backend replies with a single ready ack
|
||||
// (TranscriptLiveResponse{ready:true}); backends or models without
|
||||
// cache-aware streaming support return UNIMPLEMENTED instead. Newly
|
||||
// finalized text streams back as deltas; eou=true marks the model's
|
||||
// end-of-utterance token. One stream spans many utterances (the decoder
|
||||
// resets itself after each EOU). Closing the send side finalizes: the
|
||||
// backend flushes the decoder tail and emits a terminal message carrying
|
||||
// final_result. A second Config mid-stream resets the decode session.
|
||||
rpc AudioTranscriptionLive(stream TranscriptLiveRequest) returns (stream TranscriptLiveResponse) {}
|
||||
rpc TTS(TTSRequest) returns (Result) {}
|
||||
rpc TTSStream(TTSRequest) returns (stream Reply) {}
|
||||
rpc SoundGeneration(SoundGenerationRequest) returns (Result) {}
|
||||
@@ -124,6 +136,10 @@ message MetricsResponse {
|
||||
message TokenClassifyRequest {
|
||||
string text = 1;
|
||||
float threshold = 2;
|
||||
// ModelIdentity names the model this request is for; see
|
||||
// PredictOptions.ModelIdentity for the full rationale. Empty means "no
|
||||
// identity supplied" and backends MUST skip the check.
|
||||
string ModelIdentity = 3;
|
||||
}
|
||||
|
||||
// TokenClassifyEntity is one detected entity span. Byte offsets are
|
||||
@@ -161,6 +177,10 @@ message ScoreRequest {
|
||||
// candidates differ in length and the consumer wants a per-token
|
||||
// measure comparable across them (PMI-style scoring).
|
||||
bool length_normalize = 4;
|
||||
// ModelIdentity names the model this request is for; see
|
||||
// PredictOptions.ModelIdentity for the full rationale. Empty means "no
|
||||
// identity supplied" and backends MUST skip the check.
|
||||
string ModelIdentity = 5;
|
||||
}
|
||||
|
||||
// CandidateScore is one row in the ScoreResponse, matching by index
|
||||
@@ -192,6 +212,10 @@ message RerankRequest {
|
||||
string query = 1;
|
||||
repeated string documents = 2;
|
||||
int32 top_n = 3;
|
||||
// ModelIdentity names the model this request is for; see
|
||||
// PredictOptions.ModelIdentity for the full rationale. Empty means "no
|
||||
// identity supplied" and backends MUST skip the check.
|
||||
string ModelIdentity = 4;
|
||||
}
|
||||
|
||||
message RerankResult {
|
||||
@@ -303,6 +327,39 @@ message PredictOptions {
|
||||
int32 TopLogprobs = 51; // Number of top logprobs to return per token (maps to OpenAI top_logprobs parameter)
|
||||
map<string, string> Metadata = 52; // Generic per-request metadata (e.g., enable_thinking)
|
||||
float MinP = 53; // Minimum probability sampling threshold (0.0 = disabled)
|
||||
|
||||
// ModelIdentity names the model this request is for, so a backend can reject
|
||||
// a request that reached it by mistake instead of answering from whatever
|
||||
// model it happens to hold. In distributed mode a worker can recycle a
|
||||
// stopped backend's gRPC port for a different model's backend, and a
|
||||
// liveness-only health probe cannot tell that apart from a valid cached
|
||||
// route (#10952).
|
||||
//
|
||||
// The value is the controller's ModelConfig.Model, the SAME expression that
|
||||
// produces ModelOptions.Model at LoadModel time, so the two are equal by
|
||||
// construction rather than by convention.
|
||||
//
|
||||
// Empty means "no identity supplied": backends MUST skip the check. That
|
||||
// keeps an old controller talking to a new backend working, and covers
|
||||
// callers that legitimately synthesize a PredictOptions internally.
|
||||
//
|
||||
// Do NOT reuse TTSRequest.model or SoundGenerationRequest.model for this
|
||||
// purpose. FileStagingClient already rewrites those to worker-local absolute
|
||||
// paths (core/services/nodes/file_staging_client.go), so in distributed mode
|
||||
// they already differ from the load-time value and comparing them would
|
||||
// reject valid requests. Extending identity to those RPCs needs a separate
|
||||
// field carrying the untranslated value - which is exactly what
|
||||
// TTSRequest.ModelIdentity and SoundGenerationRequest.ModelIdentity are.
|
||||
//
|
||||
// Every other request message that reaches a backend through the distributed
|
||||
// router now carries the same ModelIdentity field, populated from the same
|
||||
// ModelConfig.Model. FileStagingClient rewrites Src/Dst/Voice/Model/
|
||||
// StartImage/EndImage/Audio and never ModelIdentity, so what the backend
|
||||
// compares is always what the controller sent.
|
||||
string ModelIdentity = 54;
|
||||
|
||||
// 24 was never assigned; reserve it so it is not silently reused.
|
||||
reserved 24;
|
||||
}
|
||||
|
||||
// ToolCallDelta represents an incremental tool call update from the C++ parser.
|
||||
@@ -472,6 +529,10 @@ message TranscriptRequest {
|
||||
float temperature = 8;
|
||||
repeated string timestamp_granularities = 9;
|
||||
bool stream = 10;
|
||||
// ModelIdentity names the model this request is for; see
|
||||
// PredictOptions.ModelIdentity for the full rationale. Empty means "no
|
||||
// identity supplied" and backends MUST skip the check.
|
||||
string ModelIdentity = 11;
|
||||
}
|
||||
|
||||
message TranscriptResult {
|
||||
@@ -479,6 +540,10 @@ message TranscriptResult {
|
||||
string text = 2;
|
||||
string language = 3;
|
||||
float duration = 4;
|
||||
// True when the decode ended on the model's end-of-utterance special token
|
||||
// (<EOU>/<EOB>, emitted by cache-aware streaming models such as
|
||||
// parakeet_realtime_eou_120m-v1). The marker itself is stripped from text.
|
||||
bool eou = 5;
|
||||
}
|
||||
|
||||
message TranscriptStreamResponse {
|
||||
@@ -486,6 +551,34 @@ message TranscriptStreamResponse {
|
||||
TranscriptResult final_result = 2;
|
||||
}
|
||||
|
||||
// === AudioTranscriptionLive messages =====================================
|
||||
|
||||
message TranscriptLiveRequest {
|
||||
oneof payload {
|
||||
TranscriptLiveConfig config = 1;
|
||||
TranscriptLiveAudio audio = 2;
|
||||
}
|
||||
}
|
||||
|
||||
message TranscriptLiveConfig {
|
||||
string language = 1; // "" => model default
|
||||
int32 sample_rate = 2; // 0 => 16000; backends may reject others
|
||||
map<string, string> params = 3; // backend-specific tuning
|
||||
}
|
||||
|
||||
message TranscriptLiveAudio {
|
||||
repeated float pcm = 1; // mono PCM in [-1,1] at config.sample_rate
|
||||
}
|
||||
|
||||
message TranscriptLiveResponse {
|
||||
bool ready = 1; // open ack: sent once, before any delta
|
||||
string delta = 2; // newly-finalized text since previous response
|
||||
bool eou = 3; // <EOU> fired during this feed (the user yielded the turn)
|
||||
repeated TranscriptWord words = 4; // words finalized by this feed (stream-relative ns)
|
||||
TranscriptResult final_result = 5; // terminal message only, after the send side closes
|
||||
bool eob = 6; // <EOB> fired: a backchannel ("uh-huh") ended — NOT a turn boundary
|
||||
}
|
||||
|
||||
message TranscriptWord {
|
||||
int64 start = 1;
|
||||
int64 end = 2;
|
||||
@@ -518,6 +611,10 @@ message GenerateImageRequest {
|
||||
|
||||
// Reference images for models that support them (e.g., Flux Kontext)
|
||||
repeated string ref_images = 12;
|
||||
// ModelIdentity names the model this request is for; see
|
||||
// PredictOptions.ModelIdentity for the full rationale. Empty means "no
|
||||
// identity supplied" and backends MUST skip the check.
|
||||
string ModelIdentity = 13;
|
||||
}
|
||||
|
||||
message GenerateVideoRequest {
|
||||
@@ -533,6 +630,14 @@ message GenerateVideoRequest {
|
||||
float cfg_scale = 10; // Classifier-free guidance scale
|
||||
int32 step = 11; // Number of inference steps
|
||||
string dst = 12; // Output path for the generated video
|
||||
string audio = 13; // Path to staged audio for audio-conditioned video
|
||||
// Backend-specific per-request generation parameters. Values are strings
|
||||
// and are validated/coerced by the selected backend.
|
||||
map<string, string> params = 14;
|
||||
// ModelIdentity names the model this request is for; see
|
||||
// PredictOptions.ModelIdentity for the full rationale. Empty means "no
|
||||
// identity supplied" and backends MUST skip the check.
|
||||
string ModelIdentity = 15;
|
||||
}
|
||||
|
||||
message TTSRequest {
|
||||
@@ -550,10 +655,26 @@ message TTSRequest {
|
||||
// (e.g. Chatterbox exaggeration/cfg_weight/temperature). Values are strings and
|
||||
// coerced by the backend; unset leaves the backend's configured defaults.
|
||||
map<string, string> params = 7;
|
||||
// ModelIdentity is a SEPARATE field from `model` above and carries the
|
||||
// UNTRANSLATED controller-side ModelConfig.Model, so a backend can reject a
|
||||
// request that reached it through a stale distributed route (#10952).
|
||||
//
|
||||
// `model` cannot be reused for this: FileStagingClient.TTS/.TTSStream and the
|
||||
// SoundGeneration path rewrite it into a worker-local absolute path
|
||||
// (core/services/nodes/file_staging_client.go), while the load-time value is
|
||||
// untranslated. In distributed mode - exactly the configuration this guards -
|
||||
// the two already differ, so comparing them would reject valid requests.
|
||||
//
|
||||
// Empty means "no identity supplied" and backends MUST skip the check.
|
||||
string ModelIdentity = 8;
|
||||
}
|
||||
|
||||
message VADRequest {
|
||||
repeated float audio = 1;
|
||||
// ModelIdentity names the model this request is for; see
|
||||
// PredictOptions.ModelIdentity for the full rationale. Empty means "no
|
||||
// identity supplied" and backends MUST skip the check.
|
||||
string ModelIdentity = 2;
|
||||
}
|
||||
|
||||
message VADSegment {
|
||||
@@ -585,6 +706,10 @@ message DiarizeRequest {
|
||||
float min_duration_on = 8; // discard segments shorter than this (seconds); 0 = backend default
|
||||
float min_duration_off = 9; // merge gaps shorter than this (seconds); 0 = backend default
|
||||
bool include_text = 10; // when the backend can emit per-segment transcript for free, ask it to populate `text`
|
||||
// ModelIdentity names the model this request is for; see
|
||||
// PredictOptions.ModelIdentity for the full rationale. Empty means "no
|
||||
// identity supplied" and backends MUST skip the check.
|
||||
string ModelIdentity = 11;
|
||||
}
|
||||
|
||||
message DiarizeSegment {
|
||||
@@ -619,6 +744,18 @@ message SoundGenerationRequest {
|
||||
optional string language = 14;
|
||||
optional string timesignature = 15;
|
||||
optional bool instrumental = 17;
|
||||
// ModelIdentity is a SEPARATE field from `model` above and carries the
|
||||
// UNTRANSLATED controller-side ModelConfig.Model, so a backend can reject a
|
||||
// request that reached it through a stale distributed route (#10952).
|
||||
//
|
||||
// `model` cannot be reused for this: FileStagingClient.TTS/.TTSStream and the
|
||||
// SoundGeneration path rewrite it into a worker-local absolute path
|
||||
// (core/services/nodes/file_staging_client.go), while the load-time value is
|
||||
// untranslated. In distributed mode - exactly the configuration this guards -
|
||||
// the two already differ, so comparing them would reject valid requests.
|
||||
//
|
||||
// Empty means "no identity supplied" and backends MUST skip the check.
|
||||
string ModelIdentity = 18;
|
||||
}
|
||||
|
||||
message TokenizationResponse {
|
||||
@@ -658,6 +795,10 @@ message DetectOptions {
|
||||
repeated float points = 3; // Point coordinates as [x1, y1, label1, x2, y2, label2, ...] (label: 1=pos, 0=neg)
|
||||
repeated float boxes = 4; // Box coordinates as [x1, y1, x2, y2, ...]
|
||||
float threshold = 5; // Detection confidence threshold
|
||||
// ModelIdentity names the model this request is for; see
|
||||
// PredictOptions.ModelIdentity for the full rationale. Empty means "no
|
||||
// identity supplied" and backends MUST skip the check.
|
||||
string ModelIdentity = 6;
|
||||
}
|
||||
|
||||
message Detection {
|
||||
@@ -680,6 +821,10 @@ message SoundDetectionRequest {
|
||||
string src = 1; // audio file path (LocalAI writes the upload to disk)
|
||||
int32 top_k = 2; // number of top tags to return (0 = all classes)
|
||||
float threshold = 3; // optional: drop tags scoring below this
|
||||
// ModelIdentity names the model this request is for; see
|
||||
// PredictOptions.ModelIdentity for the full rationale. Empty means "no
|
||||
// identity supplied" and backends MUST skip the check.
|
||||
string ModelIdentity = 4;
|
||||
}
|
||||
|
||||
message SoundClass {
|
||||
@@ -704,6 +849,10 @@ message DepthRequest {
|
||||
bool include_points = 7; // back-project to a 3D point cloud (DualDPT)
|
||||
float points_conf_thresh = 8; // keep points with confidence >= this threshold
|
||||
repeated string exports = 9; // requested exports: "glb", "colmap"
|
||||
// ModelIdentity names the model this request is for; see
|
||||
// PredictOptions.ModelIdentity for the full rationale. Empty means "no
|
||||
// identity supplied" and backends MUST skip the check.
|
||||
string ModelIdentity = 10;
|
||||
}
|
||||
|
||||
message DepthResponse {
|
||||
@@ -735,6 +884,10 @@ message FaceVerifyRequest {
|
||||
string img2 = 2; // base64-encoded image
|
||||
float threshold = 3; // cosine-distance threshold; 0 = use backend default
|
||||
bool anti_spoofing = 4; // run MiniFASNet liveness on each image; failed liveness forces verified=false
|
||||
// ModelIdentity names the model this request is for; see
|
||||
// PredictOptions.ModelIdentity for the full rationale. Empty means "no
|
||||
// identity supplied" and backends MUST skip the check.
|
||||
string ModelIdentity = 5;
|
||||
}
|
||||
|
||||
message FaceVerifyResponse {
|
||||
@@ -756,6 +909,10 @@ message FaceAnalyzeRequest {
|
||||
string img = 1; // base64-encoded image
|
||||
repeated string actions = 2; // subset of ["age","gender","emotion","race"]; empty = all-supported
|
||||
bool anti_spoofing = 3;
|
||||
// ModelIdentity names the model this request is for; see
|
||||
// PredictOptions.ModelIdentity for the full rationale. Empty means "no
|
||||
// identity supplied" and backends MUST skip the check.
|
||||
string ModelIdentity = 4;
|
||||
}
|
||||
|
||||
message FaceAnalysis {
|
||||
@@ -788,6 +945,10 @@ message VoiceVerifyRequest {
|
||||
string audio2 = 2; // path to second audio clip
|
||||
float threshold = 3; // cosine-distance threshold; 0 = use backend default
|
||||
bool anti_spoofing = 4; // reserved for future AASIST bolt-on
|
||||
// ModelIdentity names the model this request is for; see
|
||||
// PredictOptions.ModelIdentity for the full rationale. Empty means "no
|
||||
// identity supplied" and backends MUST skip the check.
|
||||
string ModelIdentity = 5;
|
||||
}
|
||||
|
||||
message VoiceVerifyResponse {
|
||||
@@ -802,6 +963,10 @@ message VoiceVerifyResponse {
|
||||
message VoiceAnalyzeRequest {
|
||||
string audio = 1; // path to audio clip
|
||||
repeated string actions = 2; // subset of ["age","gender","emotion"]; empty = all-supported
|
||||
// ModelIdentity names the model this request is for; see
|
||||
// PredictOptions.ModelIdentity for the full rationale. Empty means "no
|
||||
// identity supplied" and backends MUST skip the check.
|
||||
string ModelIdentity = 3;
|
||||
}
|
||||
|
||||
message VoiceAnalysis {
|
||||
@@ -820,6 +985,10 @@ message VoiceAnalyzeResponse {
|
||||
|
||||
message VoiceEmbedRequest {
|
||||
string audio = 1; // path to audio clip
|
||||
// ModelIdentity names the model this request is for; see
|
||||
// PredictOptions.ModelIdentity for the full rationale. Empty means "no
|
||||
// identity supplied" and backends MUST skip the check.
|
||||
string ModelIdentity = 2;
|
||||
}
|
||||
|
||||
message VoiceEmbedResponse {
|
||||
@@ -914,6 +1083,10 @@ message AudioTransformRequest {
|
||||
string reference_path = 2; // optional auxiliary; empty => zero-fill
|
||||
string dst = 3; // required, output file path
|
||||
map<string, string> params = 4; // backend-specific tuning
|
||||
// ModelIdentity names the model this request is for; see
|
||||
// PredictOptions.ModelIdentity for the full rationale. Empty means "no
|
||||
// identity supplied" and backends MUST skip the check.
|
||||
string ModelIdentity = 5;
|
||||
}
|
||||
|
||||
message AudioTransformResult {
|
||||
@@ -1212,4 +1385,3 @@ message ForwardReply {
|
||||
repeated ForwardHeader headers = 2;
|
||||
bytes body_chunk = 3;
|
||||
}
|
||||
|
||||
|
||||
105
backend/cpp/bonsai/Makefile
Normal file
105
backend/cpp/bonsai/Makefile
Normal file
@@ -0,0 +1,105 @@
|
||||
|
||||
# Pinned to the HEAD of the `prism` branch on https://github.com/PrismML-Eng/llama.cpp.
|
||||
# Auto-bumped nightly by .github/workflows/bump_deps.yaml.
|
||||
BONSAI_VERSION?=9fcaed763ccda38ea81068ad9d7f991aaddca451
|
||||
LLAMA_REPO?=https://github.com/PrismML-Eng/llama.cpp
|
||||
|
||||
CMAKE_ARGS?=
|
||||
BUILD_TYPE?=
|
||||
NATIVE?=false
|
||||
ONEAPI_VARS?=/opt/intel/oneapi/setvars.sh
|
||||
TARGET?=--target grpc-server
|
||||
JOBS?=$(shell nproc 2>/dev/null || sysctl -n hw.ncpu 2>/dev/null || echo 1)
|
||||
ARCH?=$(shell uname -m)
|
||||
|
||||
CURRENT_MAKEFILE_DIR := $(dir $(abspath $(lastword $(MAKEFILE_LIST))))
|
||||
LLAMA_CPP_DIR := $(CURRENT_MAKEFILE_DIR)/../llama-cpp
|
||||
|
||||
GREEN := \033[0;32m
|
||||
RESET := \033[0m
|
||||
|
||||
# bonsai is a llama.cpp fork (PrismML) adding the Q1_0 (1-bit) and Q2_0 (ternary)
|
||||
# weight-quantization kernels that the Bonsai / Ternary-Bonsai models ship in. Rather
|
||||
# than duplicating grpc-server.cpp / CMakeLists.txt / prepare.sh we reuse the ones in
|
||||
# backend/cpp/llama-cpp, and only swap which repo+sha the fetch step pulls. Each flavor
|
||||
# target copies ../llama-cpp into a sibling ../bonsai-<flavor>-build directory, then
|
||||
# invokes llama-cpp's own build with LLAMA_REPO/LLAMA_VERSION overridden to point at the
|
||||
# fork.
|
||||
#
|
||||
# The Q1_0/Q2_0 additions are model *weight* types decoded inside libllama, transparent
|
||||
# to the reused gRPC server, so (unlike turboquant's KV-cache types) no grpc-server.cpp
|
||||
# allow-list patch is needed. The fork branched from upstream before a few API changes
|
||||
# the shared grpc-server.cpp depends on; those are carried as patch files under
|
||||
# backend/cpp/bonsai/patches/ and applied to the cloned fork by apply-patches.sh.
|
||||
PATCHES_DIR := $(CURRENT_MAKEFILE_DIR)/patches
|
||||
|
||||
define bonsai-build
|
||||
rm -rf $(CURRENT_MAKEFILE_DIR)/../bonsai-$(1)-build
|
||||
cp -rf $(LLAMA_CPP_DIR) $(CURRENT_MAKEFILE_DIR)/../bonsai-$(1)-build
|
||||
# Drop patches vendored for upstream llama.cpp: the fork tree diverges, so
|
||||
# they reject there. Fork-specific patches live in backend/cpp/bonsai/patches/
|
||||
# and are applied by apply-patches.sh below.
|
||||
rm -rf $(CURRENT_MAKEFILE_DIR)/../bonsai-$(1)-build/patches
|
||||
$(MAKE) -C $(CURRENT_MAKEFILE_DIR)/../bonsai-$(1)-build purge
|
||||
$(info $(GREEN)I bonsai build info:$(1)$(RESET))
|
||||
LLAMA_REPO=$(LLAMA_REPO) LLAMA_VERSION=$(BONSAI_VERSION) \
|
||||
$(MAKE) -C $(CURRENT_MAKEFILE_DIR)/../bonsai-$(1)-build llama.cpp
|
||||
bash $(CURRENT_MAKEFILE_DIR)/apply-patches.sh $(CURRENT_MAKEFILE_DIR)/../bonsai-$(1)-build/llama.cpp $(PATCHES_DIR)
|
||||
CMAKE_ARGS="$(CMAKE_ARGS) $(2)" TARGET="$(3)" \
|
||||
LLAMA_REPO=$(LLAMA_REPO) LLAMA_VERSION=$(BONSAI_VERSION) \
|
||||
$(MAKE) -C $(CURRENT_MAKEFILE_DIR)/../bonsai-$(1)-build grpc-server
|
||||
cp -rfv $(CURRENT_MAKEFILE_DIR)/../bonsai-$(1)-build/grpc-server bonsai-$(1)
|
||||
endef
|
||||
|
||||
bonsai-avx2:
|
||||
$(call bonsai-build,avx2,-DGGML_AVX=on -DGGML_AVX2=on -DGGML_AVX512=off -DGGML_FMA=on -DGGML_F16C=on,--target grpc-server)
|
||||
|
||||
bonsai-avx512:
|
||||
$(call bonsai-build,avx512,-DGGML_AVX=on -DGGML_AVX2=off -DGGML_AVX512=on -DGGML_FMA=on -DGGML_F16C=on,--target grpc-server)
|
||||
|
||||
bonsai-avx:
|
||||
$(call bonsai-build,avx,-DGGML_AVX=on -DGGML_AVX2=off -DGGML_AVX512=off -DGGML_FMA=off -DGGML_F16C=off -DGGML_BMI2=off,--target grpc-server)
|
||||
|
||||
bonsai-fallback:
|
||||
$(call bonsai-build,fallback,-DGGML_AVX=off -DGGML_AVX2=off -DGGML_AVX512=off -DGGML_FMA=off -DGGML_F16C=off -DGGML_BMI2=off,--target grpc-server)
|
||||
|
||||
# Single-build CPU backend via ggml CPU_ALL_VARIANTS (mirrors llama-cpp-cpu-all).
|
||||
# bonsai reuses backend/cpp/llama-cpp's CMakeLists.txt (hw_grpc_proto STATIC) and
|
||||
# Makefile (SHARED_LIBS make-var + EXTRA_CMAKE_ARGS), so this passes the same overrides
|
||||
# through to the copied build: SHARED_LIBS=ON, the DL flags, and --target ggml (which
|
||||
# pulls in the per-microarch libggml-cpu-*.so via ggml's add_dependencies). The .so set
|
||||
# is collected for package.sh to bundle into package/lib.
|
||||
bonsai-cpu-all:
|
||||
rm -rf $(CURRENT_MAKEFILE_DIR)/../bonsai-cpu-all-build
|
||||
cp -rf $(LLAMA_CPP_DIR) $(CURRENT_MAKEFILE_DIR)/../bonsai-cpu-all-build
|
||||
# Drop patches vendored for upstream llama.cpp: the fork tree diverges, so
|
||||
# they reject there. Fork-specific patches live in backend/cpp/bonsai/patches/
|
||||
# and are applied by apply-patches.sh below.
|
||||
rm -rf $(CURRENT_MAKEFILE_DIR)/../bonsai-cpu-all-build/patches
|
||||
$(MAKE) -C $(CURRENT_MAKEFILE_DIR)/../bonsai-cpu-all-build purge
|
||||
$(info $(GREEN)I bonsai build info:cpu-all-variants$(RESET))
|
||||
LLAMA_REPO=$(LLAMA_REPO) LLAMA_VERSION=$(BONSAI_VERSION) \
|
||||
$(MAKE) -C $(CURRENT_MAKEFILE_DIR)/../bonsai-cpu-all-build llama.cpp
|
||||
bash $(CURRENT_MAKEFILE_DIR)/apply-patches.sh $(CURRENT_MAKEFILE_DIR)/../bonsai-cpu-all-build/llama.cpp $(PATCHES_DIR)
|
||||
SHARED_LIBS=ON EXTRA_CMAKE_ARGS="-DGGML_BACKEND_DL=ON -DGGML_CPU_ALL_VARIANTS=ON" TARGET="--target grpc-server --target ggml" \
|
||||
LLAMA_REPO=$(LLAMA_REPO) LLAMA_VERSION=$(BONSAI_VERSION) \
|
||||
$(MAKE) -C $(CURRENT_MAKEFILE_DIR)/../bonsai-cpu-all-build grpc-server
|
||||
cp -rfv $(CURRENT_MAKEFILE_DIR)/../bonsai-cpu-all-build/grpc-server bonsai-cpu-all
|
||||
rm -rf ggml-shared-libs && mkdir -p ggml-shared-libs
|
||||
find $(CURRENT_MAKEFILE_DIR)/../bonsai-cpu-all-build/llama.cpp/build \( -name '*.so*' -o -name '*.dylib' \) -exec cp -av {} ggml-shared-libs/ \;
|
||||
@echo "Collected ggml shared backends:" && ls -la ggml-shared-libs/
|
||||
|
||||
bonsai-grpc:
|
||||
$(call bonsai-build,grpc,-DGGML_RPC=ON -DGGML_AVX=off -DGGML_AVX2=off -DGGML_AVX512=off -DGGML_FMA=off -DGGML_F16C=off -DGGML_BMI2=off,--target grpc-server --target rpc-server)
|
||||
|
||||
bonsai-rpc-server: bonsai-grpc
|
||||
cp -rf $(CURRENT_MAKEFILE_DIR)/../bonsai-grpc-build/llama.cpp/build/bin/rpc-server bonsai-rpc-server
|
||||
|
||||
package:
|
||||
bash package.sh
|
||||
|
||||
purge:
|
||||
rm -rf $(CURRENT_MAKEFILE_DIR)/../bonsai-*-build
|
||||
rm -rf bonsai-* package
|
||||
|
||||
clean: purge
|
||||
48
backend/cpp/bonsai/apply-patches.sh
Executable file
48
backend/cpp/bonsai/apply-patches.sh
Executable file
@@ -0,0 +1,48 @@
|
||||
#!/bin/bash
|
||||
# Apply the bonsai patch series to a cloned PrismML llama.cpp (prism branch) checkout.
|
||||
#
|
||||
# The prism fork branched from upstream llama.cpp before a number of API changes that the
|
||||
# shared backend/cpp/llama-cpp/grpc-server.cpp depends on. We carry those upstream commits
|
||||
# as patch files under backend/cpp/bonsai/patches/ and apply them here so the reused
|
||||
# grpc-server source compiles against the fork unmodified.
|
||||
#
|
||||
# Drop the corresponding patch from patches/ whenever the fork catches up with upstream —
|
||||
# the build will fail fast if a patch stops applying, which is the signal to retire it.
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
if [[ $# -ne 2 ]]; then
|
||||
echo "usage: $0 <llama.cpp-src-dir> <patches-dir>" >&2
|
||||
exit 2
|
||||
fi
|
||||
|
||||
SRC_DIR=$1
|
||||
PATCHES_DIR=$2
|
||||
|
||||
if [[ ! -d "$SRC_DIR" ]]; then
|
||||
echo "source dir does not exist: $SRC_DIR" >&2
|
||||
exit 2
|
||||
fi
|
||||
|
||||
if [[ ! -d "$PATCHES_DIR" ]]; then
|
||||
echo "no patches dir at $PATCHES_DIR, nothing to apply"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
shopt -s nullglob
|
||||
patches=("$PATCHES_DIR"/*.patch)
|
||||
shopt -u nullglob
|
||||
|
||||
if [[ ${#patches[@]} -eq 0 ]]; then
|
||||
echo "no .patch files in $PATCHES_DIR, nothing to apply"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
cd "$SRC_DIR"
|
||||
|
||||
for patch in "${patches[@]}"; do
|
||||
echo "==> applying $patch"
|
||||
git apply --verbose "$patch"
|
||||
done
|
||||
|
||||
echo "all bonsai patches applied successfully"
|
||||
@@ -11,7 +11,7 @@ REPO_ROOT="${CURDIR}/../../.."
|
||||
# Create lib directory
|
||||
mkdir -p $CURDIR/package/lib
|
||||
|
||||
cp -avrf $CURDIR/llama-cpp-localai-paged-* $CURDIR/package/
|
||||
cp -avrf $CURDIR/bonsai-* $CURDIR/package/
|
||||
cp -rfv $CURDIR/run.sh $CURDIR/package/
|
||||
|
||||
# Bundle the ggml shared backends from the CPU_ALL_VARIANTS build into package/lib. ggml
|
||||
19
backend/cpp/bonsai/patches/README.md
Normal file
19
backend/cpp/bonsai/patches/README.md
Normal file
@@ -0,0 +1,19 @@
|
||||
# bonsai fork skew patches
|
||||
|
||||
The `bonsai` backend reuses `backend/cpp/llama-cpp/grpc-server.cpp` (written against
|
||||
LocalAI's pinned *upstream* llama.cpp) but compiles it against the PrismML `prism` fork,
|
||||
which branched from upstream some commits earlier. Any upstream API change that the shared
|
||||
gRPC server depends on, but that the fork does not yet carry, is back-ported here as a
|
||||
`*.patch` file and applied to the cloned fork checkout by `../apply-patches.sh`.
|
||||
|
||||
CI treats both this directory and `backend/cpp/llama-cpp/` as Bonsai inputs, since
|
||||
the wrapper copies and builds the shared llama.cpp backend sources.
|
||||
|
||||
Rules:
|
||||
|
||||
- One upstream commit (or minimal hunk) per patch, named `NNNN-short-description.patch`.
|
||||
- Patches are applied with `git apply` from the fork's checkout root.
|
||||
- `apply-patches.sh` fails fast if a patch stops applying cleanly — that is the signal the
|
||||
fork has caught up (or diverged), so re-cut or drop the patch.
|
||||
- Keep this set as small as possible; the long-term fix is the fork rebasing onto a newer
|
||||
upstream (or Q1_0/Q2_0 landing in mainline llama.cpp, retiring this backend entirely).
|
||||
56
backend/cpp/bonsai/run.sh
Executable file
56
backend/cpp/bonsai/run.sh
Executable file
@@ -0,0 +1,56 @@
|
||||
#!/bin/bash
|
||||
set -ex
|
||||
|
||||
# Get the absolute current dir where the script is located
|
||||
CURDIR=$(dirname "$(realpath "$0")")
|
||||
|
||||
cd /
|
||||
|
||||
echo "CPU info:"
|
||||
grep -e "model\sname" /proc/cpuinfo | head -1
|
||||
grep -e "flags" /proc/cpuinfo | head -1
|
||||
|
||||
BINARY=bonsai-fallback
|
||||
|
||||
# x86/arm64 ship a single bonsai-cpu-all built with ggml CPU_ALL_VARIANTS: ggml's
|
||||
# backend registry dlopens the best libggml-cpu-*.so for this host, so no shell-side
|
||||
# probing. ROCm ships only bonsai-fallback, so fall back to it when cpu-all is absent.
|
||||
if [ -e "$CURDIR"/bonsai-cpu-all ]; then
|
||||
BINARY=bonsai-cpu-all
|
||||
fi
|
||||
|
||||
if [ -n "$LLAMACPP_GRPC_SERVERS" ]; then
|
||||
if [ -e "$CURDIR"/bonsai-grpc ]; then
|
||||
BINARY=bonsai-grpc
|
||||
fi
|
||||
fi
|
||||
|
||||
# Extend ld library path with the dir where this script is located/lib
|
||||
if [ "$(uname)" == "Darwin" ]; then
|
||||
export DYLD_LIBRARY_PATH="$CURDIR"/lib:$DYLD_LIBRARY_PATH
|
||||
else
|
||||
export LD_LIBRARY_PATH="$CURDIR"/lib:$LD_LIBRARY_PATH
|
||||
# Tell rocBLAS where to find TensileLibrary data (GPU kernel tuning files)
|
||||
if [ -d "$CURDIR/lib/rocblas/library" ]; then
|
||||
export ROCBLAS_TENSILE_LIBPATH="$CURDIR"/lib/rocblas/library
|
||||
fi
|
||||
# Same for hipBLASLt (rocblaslt): the bundled libhipblaslt.so resolves its
|
||||
# TensileLibrary_lazy_gfx*.dat kernel data relative to itself, so point it at
|
||||
# the bundled data or it falls back to slow generic kernels (issue #10660).
|
||||
if [ -d "$CURDIR/lib/hipblaslt/library" ]; then
|
||||
export HIPBLASLT_TENSILE_LIBPATH="$CURDIR"/lib/hipblaslt/library
|
||||
fi
|
||||
fi
|
||||
|
||||
# If there is a lib/ld.so, use it
|
||||
if [ -f "$CURDIR"/lib/ld.so ]; then
|
||||
echo "Using lib/ld.so"
|
||||
echo "Using binary: $BINARY"
|
||||
exec "$CURDIR"/lib/ld.so "$CURDIR"/$BINARY "$@"
|
||||
fi
|
||||
|
||||
echo "Using binary: $BINARY"
|
||||
exec "$CURDIR"/$BINARY "$@"
|
||||
|
||||
# We should never reach this point, however just in case we do, run fallback
|
||||
exec "$CURDIR"/bonsai-fallback "$@"
|
||||
@@ -51,6 +51,11 @@ namespace {
|
||||
|
||||
// Global state - ds4 is single-engine-per-process by design.
|
||||
std::mutex g_engine_mu;
|
||||
// The ModelOptions.Model this process loaded, compared against
|
||||
// PredictOptions.ModelIdentity so a request that arrived through a stale
|
||||
// distributed route is rejected rather than answered from the wrong model
|
||||
// (#10952). Guarded by g_engine_mu like the rest of the engine state.
|
||||
std::string g_loaded_model_identity;
|
||||
ds4_engine *g_engine = nullptr;
|
||||
ds4_session *g_session = nullptr;
|
||||
int g_ctx_size = 32768;
|
||||
@@ -562,6 +567,24 @@ static void build_prompt(ds4_engine *engine, const backend::PredictOptions *requ
|
||||
ds4_chat_append_assistant_prefix(engine, out, think);
|
||||
}
|
||||
|
||||
// check_model_identity mirrors pkg/grpc/server.go and
|
||||
// backend/python/common/model_identity.py. Either side empty means "skip": the
|
||||
// request side is empty for a controller that predates the field, the loaded
|
||||
// side when such a controller performed the load. A false rejection is worse
|
||||
// than the miss it prevents. Callers must already hold g_engine_mu.
|
||||
static GStatus check_model_identity(const backend::PredictOptions *request) {
|
||||
if (request == nullptr || request->modelidentity().empty()) return GStatus::OK;
|
||||
if (g_loaded_model_identity.empty() ||
|
||||
g_loaded_model_identity == request->modelidentity()) {
|
||||
return GStatus::OK;
|
||||
}
|
||||
// NOT_FOUND plus this exact sentinel is the cross-language contract the
|
||||
// router matches on (grpcerrors.ModelMismatchSentinel).
|
||||
return GStatus(StatusCode::NOT_FOUND,
|
||||
"ds4: model identity mismatch: loaded \"" + g_loaded_model_identity +
|
||||
"\", requested \"" + request->modelidentity() + "\"");
|
||||
}
|
||||
|
||||
class DS4Backend final : public backend::Backend::Service {
|
||||
public:
|
||||
GStatus Health(ServerContext *, const backend::HealthMessage *,
|
||||
@@ -716,6 +739,7 @@ public:
|
||||
}
|
||||
|
||||
result->set_success(true);
|
||||
g_loaded_model_identity = request->model();
|
||||
result->set_message("loaded " + model_path);
|
||||
return GStatus::OK;
|
||||
}
|
||||
@@ -724,6 +748,7 @@ public:
|
||||
backend::TokenizationResponse *response) override {
|
||||
std::lock_guard<std::mutex> lock(g_engine_mu);
|
||||
if (!g_engine) return GStatus(StatusCode::FAILED_PRECONDITION, "ds4: model not loaded");
|
||||
if (GStatus id = check_model_identity(request); !id.ok()) return id;
|
||||
ds4_tokens out = {};
|
||||
ds4_tokenize_text(g_engine, request->prompt().c_str(), &out);
|
||||
for (int i = 0; i < out.len; ++i) response->add_tokens(out.v[i]);
|
||||
@@ -738,6 +763,7 @@ public:
|
||||
if (!g_engine || !g_session) {
|
||||
return GStatus(StatusCode::FAILED_PRECONDITION, "ds4: model not loaded");
|
||||
}
|
||||
if (GStatus id = check_model_identity(request); !id.ok()) return id;
|
||||
if (std::string route_err = wait_route_ready(lock); !route_err.empty()) {
|
||||
return GStatus(StatusCode::UNAVAILABLE, route_err);
|
||||
}
|
||||
@@ -837,6 +863,7 @@ public:
|
||||
if (!g_engine || !g_session) {
|
||||
return GStatus(StatusCode::FAILED_PRECONDITION, "ds4: model not loaded");
|
||||
}
|
||||
if (GStatus id = check_model_identity(request); !id.ok()) return id;
|
||||
if (std::string route_err = wait_route_ready(lock); !route_err.empty()) {
|
||||
return GStatus(StatusCode::UNAVAILABLE, route_err);
|
||||
}
|
||||
|
||||
@@ -1,12 +1,14 @@
|
||||
#!/bin/bash
|
||||
set -e
|
||||
set -euo pipefail
|
||||
CURDIR=$(dirname "$(realpath "$0")")
|
||||
REPO_ROOT="${CURDIR}/../../.."
|
||||
PACKAGE_DIR="$CURDIR/package"
|
||||
|
||||
mkdir -p "$CURDIR/package/lib"
|
||||
cp -avf "$CURDIR/grpc-server" "$CURDIR/package/"
|
||||
cp -avf "$CURDIR/ds4-worker" "$CURDIR/package/"
|
||||
cp -rfv "$CURDIR/run.sh" "$CURDIR/package/"
|
||||
rm -rf "$PACKAGE_DIR"
|
||||
mkdir -p "$PACKAGE_DIR/lib"
|
||||
cp -avf "$CURDIR/grpc-server" "$PACKAGE_DIR/"
|
||||
cp -avf "$CURDIR/ds4-worker" "$PACKAGE_DIR/"
|
||||
cp -rfv "$CURDIR/run.sh" "$PACKAGE_DIR/"
|
||||
|
||||
UNAME_S=$(uname -s)
|
||||
if [ "$UNAME_S" = "Darwin" ]; then
|
||||
@@ -16,25 +18,54 @@ if [ "$UNAME_S" = "Darwin" ]; then
|
||||
fi
|
||||
|
||||
if [ -f "/lib64/ld-linux-x86-64.so.2" ]; then
|
||||
cp -arfLv /lib64/ld-linux-x86-64.so.2 "$CURDIR/package/lib/ld.so"
|
||||
LIBDIR=/lib/x86_64-linux-gnu
|
||||
cp -arfLv /lib64/ld-linux-x86-64.so.2 "$PACKAGE_DIR/lib/ld.so"
|
||||
elif [ -f "/lib/ld-linux-aarch64.so.1" ]; then
|
||||
cp -arfLv /lib/ld-linux-aarch64.so.1 "$CURDIR/package/lib/ld.so"
|
||||
LIBDIR=/lib/aarch64-linux-gnu
|
||||
cp -arfLv /lib/ld-linux-aarch64.so.1 "$PACKAGE_DIR/lib/ld.so"
|
||||
else
|
||||
echo "package.sh: unknown architecture" >&2; exit 1
|
||||
fi
|
||||
|
||||
for lib in libc.so.6 libgcc_s.so.1 libstdc++.so.6 libm.so.6 libgomp.so.1 \
|
||||
libdl.so.2 librt.so.1 libpthread.so.0; do
|
||||
cp -arfLv "$LIBDIR/$lib" "$CURDIR/package/lib/$lib"
|
||||
# Bundle the complete dependency closure for both executables. In particular,
|
||||
# grpc-server links the distro gRPC/protobuf/absl stack; copying only the core
|
||||
# C/C++ runtime libraries leaves the scratch image unable to start.
|
||||
{
|
||||
ldd "$CURDIR/grpc-server"
|
||||
ldd "$CURDIR/ds4-worker"
|
||||
} | awk '$2 == "=>" && $3 ~ /^\// { print $3 }' | sort -u | \
|
||||
while read -r so; do
|
||||
cp -arfLv "$so" "$PACKAGE_DIR/lib/"
|
||||
done
|
||||
|
||||
GPU_LIB_SCRIPT="${REPO_ROOT}/scripts/build/package-gpu-libs.sh"
|
||||
if [ -f "$GPU_LIB_SCRIPT" ]; then
|
||||
source "$GPU_LIB_SCRIPT" "$CURDIR/package/lib"
|
||||
# shellcheck source=/dev/null
|
||||
source "$GPU_LIB_SCRIPT" "$PACKAGE_DIR/lib"
|
||||
package_gpu_libs
|
||||
fi
|
||||
|
||||
# Resolve every dependency through the same loader and library path used by
|
||||
# the from-scratch image. The loader can still search host defaults, so reject
|
||||
# any absolute dependency path that escapes the package instead of accepting a
|
||||
# false-positive validation against a library that scratch will not contain.
|
||||
validate_packaged_binary() {
|
||||
local binary="$1"
|
||||
local resolution
|
||||
resolution=$("$PACKAGE_DIR/lib/ld.so" \
|
||||
--library-path "$PACKAGE_DIR/lib" \
|
||||
--list "$PACKAGE_DIR/$binary")
|
||||
|
||||
printf '%s\n' "$resolution" | awk -v prefix="$PACKAGE_DIR/lib/" '
|
||||
$2 == "=>" && $3 ~ /^\// && index($3, prefix) != 1 {
|
||||
print "package.sh: dependency resolved outside package: " $0 > "/dev/stderr"
|
||||
invalid = 1
|
||||
}
|
||||
END { exit invalid }
|
||||
'
|
||||
}
|
||||
|
||||
for binary in grpc-server ds4-worker; do
|
||||
validate_packaged_binary "$binary"
|
||||
done
|
||||
|
||||
echo "ds4 package contents:"
|
||||
ls -lah "$CURDIR/package/" "$CURDIR/package/lib/"
|
||||
ls -lah "$PACKAGE_DIR/" "$PACKAGE_DIR/lib/"
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
|
||||
IK_LLAMA_VERSION?=f96eaddba8bed6a9a5e628bbf6a566775c70b49c
|
||||
IK_LLAMA_VERSION?=9d07d8681ece159a89fb4e16a1f9c9f3a5fac20f
|
||||
LLAMA_REPO?=https://github.com/ikawrakow/ik_llama.cpp
|
||||
|
||||
CMAKE_ARGS?=
|
||||
|
||||
@@ -2412,7 +2412,33 @@ static void params_parse(const backend::ModelOptions* request,
|
||||
|
||||
// GRPC Server start
|
||||
class BackendServiceImpl final : public backend::Backend::Service {
|
||||
private:
|
||||
// The ModelOptions.Model this process was loaded with. Compared against
|
||||
// PredictOptions.ModelIdentity so a request that reached us through a stale
|
||||
// distributed route is rejected instead of answered from the wrong model
|
||||
// (#10952).
|
||||
std::string loaded_model_identity;
|
||||
|
||||
public:
|
||||
// checkModelIdentity mirrors pkg/grpc/server.go and
|
||||
// backend/python/common/model_identity.py. Either side being empty means
|
||||
// "skip": the request side is empty for a controller that predates the field,
|
||||
// and the loaded side is empty when such a controller performed the load. A
|
||||
// false rejection is worse than the miss it prevents.
|
||||
grpc::Status checkModelIdentity(const backend::PredictOptions* request) {
|
||||
if (request == nullptr || request->modelidentity().empty()) {
|
||||
return grpc::Status::OK;
|
||||
}
|
||||
if (loaded_model_identity.empty() || loaded_model_identity == request->modelidentity()) {
|
||||
return grpc::Status::OK;
|
||||
}
|
||||
// NOT_FOUND plus this exact sentinel is the cross-language contract the
|
||||
// router matches on (grpcerrors.ModelMismatchSentinel).
|
||||
return grpc::Status(grpc::StatusCode::NOT_FOUND,
|
||||
"ik-llama-cpp: model identity mismatch: loaded \"" + loaded_model_identity +
|
||||
"\", requested \"" + request->modelidentity() + "\"");
|
||||
}
|
||||
|
||||
grpc::Status Health(ServerContext* context, const backend::HealthMessage* request, backend::Reply* reply) {
|
||||
// Implement Health RPC
|
||||
reply->set_message("OK");
|
||||
@@ -2438,9 +2464,12 @@ public:
|
||||
result->set_message("Loading succeeded");
|
||||
result->set_success(true);
|
||||
loaded_model = true;
|
||||
loaded_model_identity = request->model();
|
||||
return Status::OK;
|
||||
}
|
||||
grpc::Status PredictStream(grpc::ServerContext* context, const backend::PredictOptions* request, grpc::ServerWriter<backend::Reply>* writer) override {
|
||||
auto identity = checkModelIdentity(request);
|
||||
if (!identity.ok()) return identity;
|
||||
json data = parse_options(true, request, llama);
|
||||
const int task_id = llama.queue_tasks.get_new_id();
|
||||
llama.queue_results.add_waiting_task_id(task_id);
|
||||
@@ -2495,6 +2524,8 @@ public:
|
||||
|
||||
|
||||
grpc::Status Predict(ServerContext* context, const backend::PredictOptions* request, backend::Reply* reply) {
|
||||
auto identity = checkModelIdentity(request);
|
||||
if (!identity.ok()) return identity;
|
||||
json data = parse_options(false, request, llama);
|
||||
const int task_id = llama.queue_tasks.get_new_id();
|
||||
llama.queue_results.add_waiting_task_id(task_id);
|
||||
@@ -2532,6 +2563,8 @@ public:
|
||||
|
||||
/// https://github.com/ggerganov/llama.cpp/blob/aa2341298924ac89778252015efcb792f2df1e20/examples/server/server.cpp#L2969
|
||||
grpc::Status Embedding(ServerContext* context, const backend::PredictOptions* request, backend::EmbeddingResult* embeddingResult) {
|
||||
auto identity = checkModelIdentity(request);
|
||||
if (!identity.ok()) return identity;
|
||||
json data = parse_options(false, request, llama);
|
||||
const int task_id = llama.queue_tasks.get_new_id();
|
||||
llama.queue_results.add_waiting_task_id(task_id);
|
||||
@@ -2556,6 +2589,8 @@ public:
|
||||
}
|
||||
|
||||
grpc::Status TokenizeString(ServerContext* context, const backend::PredictOptions* request, backend::TokenizationResponse* response){
|
||||
auto identity = checkModelIdentity(request);
|
||||
if (!identity.ok()) return identity;
|
||||
json data = parse_options(false, request, llama);
|
||||
|
||||
std::vector<llama_token> tokens = llama.tokenize(data["prompt"],false);
|
||||
|
||||
@@ -1,157 +0,0 @@
|
||||
|
||||
# llama-cpp-localai-paged is LocalAI's paged-attention llama.cpp variant. It
|
||||
# builds upstream llama.cpp with the LocalAI paged-attention patch series
|
||||
# (patches/paged/, vendored in THIS backend) applied on top. It reuses
|
||||
# backend/cpp/llama-cpp's grpc-server.cpp / CMakeLists.txt / prepare.sh / Makefile
|
||||
# sources verbatim via a thin wrapper - the stock llama-cpp backend is pure
|
||||
# upstream and carries NONE of the paged patches; this backend OWNS them.
|
||||
#
|
||||
# Pin handling (mirrors the turboquant wrapper, the precedent this is modelled
|
||||
# on): the paged patch series is hand-verified bit-exact against ONE specific
|
||||
# llama.cpp tip and re-exported by the manual PIN_SYNC process
|
||||
# (README section 7 + .agents/llama-cpp-localai-paged-backend.md). A naive
|
||||
# pin bump would move the tip out from
|
||||
# under the patches and break `git apply` at build time, so this backend OWNS
|
||||
# its pin (LLAMA_VERSION below) instead of inheriting the auto-bumped stock pin
|
||||
# from backend/cpp/llama-cpp/Makefile. The override is forced into every copied
|
||||
# build via `LLAMA_VERSION=$(LLAMA_VERSION)`. There is deliberately NO
|
||||
# bump_deps.yaml entry for it: it is advanced ONLY by PIN_SYNC, never nightly.
|
||||
# (turboquant CAN auto-bump because its fork branch carries the patches; the
|
||||
# paged series is vendored as .patch files here, so it cannot.)
|
||||
#
|
||||
# - NO patch-grpc-server.sh and NO apply-patches.sh: the shared grpc-server.cpp
|
||||
# already carries the (runtime-gated) paged option hooks, and the paged patch
|
||||
# series (patches/paged/) is applied by THIS Makefile's own apply step onto
|
||||
# the freshly cloned tree, using the same strict `git apply` method the stock
|
||||
# build uses for base patches. The stock llama-cpp Makefile applies only its
|
||||
# own (currently empty) base patches/ series, never the paged one.
|
||||
|
||||
# Manually pin-synced llama.cpp tip the paged patch series is verified against.
|
||||
# Decoupled from the auto-bumped stock pin in backend/cpp/llama-cpp/Makefile so
|
||||
# the nightly llama.cpp bump cannot silently break the vendored paged patches.
|
||||
# Advance ONLY via the PIN_SYNC process (rebase patches + bit-exact gate +
|
||||
# re-export), then update this value. See:
|
||||
# README section 7 + .agents/llama-cpp-localai-paged-backend.md
|
||||
#
|
||||
# This pin = the manual, verified sync. The signal telling you WHEN to do the
|
||||
# next sync is the early-warning canary
|
||||
# (.github/workflows/llama-cpp-paged-canary.yml): weekly it applies + compiles
|
||||
# this patch series against the latest upstream llama.cpp tip and goes red the
|
||||
# moment upstream drifts past the patches. Canary red -> run a PIN_SYNC, then
|
||||
# bump this value. The canary never touches this pin; it is signal-only.
|
||||
#
|
||||
# HARD CONSTRAINT: keep this == the stock llama-cpp pin (backend/cpp/llama-cpp/
|
||||
# Makefile). grpc-server.cpp is SHARED with the stock backend and tracks the
|
||||
# stock pin; a paged pin that diverges PAST an upstream server-API refactor
|
||||
# breaks the grpc-server LINK even when the patches are byte-for-byte bit-exact.
|
||||
# The c299a92c bump did exactly this: patches applied + greedy-md5 bit-exact, but
|
||||
# grpc-server.cpp failed to link with undefined references to stream_* server
|
||||
# helpers that the refactor pulled into the headers grpc-server.cpp includes.
|
||||
# Therefore a PIN_SYNC must pass the FULL grpc-server build/link on CI, not only
|
||||
# the bit-exact gate. See README section 7 + .agents/llama-cpp-localai-paged-backend.md.
|
||||
LLAMA_VERSION?=0ed235ea2c17a19fc8238668653946721ed136fd
|
||||
|
||||
CMAKE_ARGS?=
|
||||
BUILD_TYPE?=
|
||||
NATIVE?=false
|
||||
ONEAPI_VARS?=/opt/intel/oneapi/setvars.sh
|
||||
TARGET?=--target grpc-server
|
||||
JOBS?=$(shell nproc 2>/dev/null || sysctl -n hw.ncpu 2>/dev/null || echo 1)
|
||||
ARCH?=$(shell uname -m)
|
||||
|
||||
CURRENT_MAKEFILE_DIR := $(dir $(abspath $(lastword $(MAKEFILE_LIST))))
|
||||
LLAMA_CPP_DIR := $(CURRENT_MAKEFILE_DIR)/../llama-cpp
|
||||
# OUR vendored paged-attention patch series. Owned by this backend; the stock
|
||||
# llama-cpp backend no longer carries it. Applied onto each freshly cloned
|
||||
# llama.cpp tree by apply-paged-patches below (strict git apply).
|
||||
PAGED_PATCHES_DIR := $(CURRENT_MAKEFILE_DIR)/patches/paged
|
||||
|
||||
GREEN := \033[0;32m
|
||||
RESET := \033[0m
|
||||
|
||||
# Apply OUR vendored paged-attention patch series (patches/paged/0*.patch) onto a
|
||||
# freshly cloned llama.cpp tree ($(1)) using the SAME strict git-apply method the
|
||||
# stock build uses for its base patches (backend/cpp/llama-cpp/Makefile `llama.cpp`
|
||||
# target). Strict: any patch that no longer applies aborts the build (exit 1) -
|
||||
# that is the signal to run a PIN_SYNC, never to bump the pin blindly. The series
|
||||
# is owned by THIS backend, not by the now-pure stock llama-cpp backend.
|
||||
define apply-paged-patches
|
||||
cd $(1) && \
|
||||
for p in $(PAGED_PATCHES_DIR)/0*.patch; do \
|
||||
[ -e "$$p" ] || continue; \
|
||||
echo "applying llama.cpp PAGED patch: $$p"; \
|
||||
git apply --verbose "$$p" || { echo "paged patch failed: $$p"; exit 1; }; \
|
||||
done
|
||||
endef
|
||||
|
||||
# Each flavor target:
|
||||
# 1. copies backend/cpp/llama-cpp/ (grpc-server.cpp + prepare.sh +
|
||||
# CMakeLists.txt + Makefile) into a sibling
|
||||
# llama-cpp-localai-paged-<flavor>-build directory;
|
||||
# 2. clones OUR pinned upstream llama.cpp into that copy via the copy's own
|
||||
# `llama.cpp` target (which applies the stock base patches/ series, normally
|
||||
# empty), then applies THIS backend's paged patch series (patches/paged/)
|
||||
# onto the cloned tree with strict `git apply` (apply-paged-patches);
|
||||
# 3. runs the copy's `grpc-server` target and copies the produced binary up as
|
||||
# llama-cpp-localai-paged-<flavor>.
|
||||
# We clone+patch only the *copy*, never the original under backend/cpp/llama-cpp/,
|
||||
# so the stock llama-cpp build stays untouched and patch-free.
|
||||
define paged-build
|
||||
rm -rf $(CURRENT_MAKEFILE_DIR)/../llama-cpp-localai-paged-$(1)-build
|
||||
cp -rf $(LLAMA_CPP_DIR) $(CURRENT_MAKEFILE_DIR)/../llama-cpp-localai-paged-$(1)-build
|
||||
$(MAKE) -C $(CURRENT_MAKEFILE_DIR)/../llama-cpp-localai-paged-$(1)-build purge
|
||||
$(info $(GREEN)I llama-cpp-localai-paged build info:$(1)$(RESET))
|
||||
LLAMA_VERSION=$(LLAMA_VERSION) $(MAKE) -C $(CURRENT_MAKEFILE_DIR)/../llama-cpp-localai-paged-$(1)-build llama.cpp
|
||||
$(call apply-paged-patches,$(CURRENT_MAKEFILE_DIR)/../llama-cpp-localai-paged-$(1)-build/llama.cpp)
|
||||
CMAKE_ARGS="$(CMAKE_ARGS) $(2)" TARGET="$(3)" LLAMA_VERSION=$(LLAMA_VERSION) \
|
||||
$(MAKE) -C $(CURRENT_MAKEFILE_DIR)/../llama-cpp-localai-paged-$(1)-build grpc-server
|
||||
cp -rfv $(CURRENT_MAKEFILE_DIR)/../llama-cpp-localai-paged-$(1)-build/grpc-server llama-cpp-localai-paged-$(1)
|
||||
endef
|
||||
|
||||
llama-cpp-localai-paged-avx2:
|
||||
$(call paged-build,avx2,-DGGML_AVX=on -DGGML_AVX2=on -DGGML_AVX512=off -DGGML_FMA=on -DGGML_F16C=on,--target grpc-server)
|
||||
|
||||
llama-cpp-localai-paged-avx512:
|
||||
$(call paged-build,avx512,-DGGML_AVX=on -DGGML_AVX2=off -DGGML_AVX512=on -DGGML_FMA=on -DGGML_F16C=on,--target grpc-server)
|
||||
|
||||
llama-cpp-localai-paged-avx:
|
||||
$(call paged-build,avx,-DGGML_AVX=on -DGGML_AVX2=off -DGGML_AVX512=off -DGGML_FMA=off -DGGML_F16C=off -DGGML_BMI2=off,--target grpc-server)
|
||||
|
||||
llama-cpp-localai-paged-fallback:
|
||||
$(call paged-build,fallback,-DGGML_AVX=off -DGGML_AVX2=off -DGGML_AVX512=off -DGGML_FMA=off -DGGML_F16C=off -DGGML_BMI2=off,--target grpc-server)
|
||||
|
||||
# Single-build CPU backend via ggml CPU_ALL_VARIANTS (mirrors llama-cpp-cpu-all).
|
||||
# Reuses backend/cpp/llama-cpp's CMakeLists.txt (hw_grpc_proto STATIC) and
|
||||
# Makefile (SHARED_LIBS make-var + EXTRA_CMAKE_ARGS), so this passes the same
|
||||
# overrides through to the copied build: SHARED_LIBS=ON, the DL flags, and
|
||||
# --target ggml (which pulls in the per-microarch libggml-cpu-*.so via ggml's
|
||||
# add_dependencies). The .so set is collected for package.sh to bundle into
|
||||
# package/lib.
|
||||
llama-cpp-localai-paged-cpu-all:
|
||||
rm -rf $(CURRENT_MAKEFILE_DIR)/../llama-cpp-localai-paged-cpu-all-build
|
||||
cp -rf $(LLAMA_CPP_DIR) $(CURRENT_MAKEFILE_DIR)/../llama-cpp-localai-paged-cpu-all-build
|
||||
$(MAKE) -C $(CURRENT_MAKEFILE_DIR)/../llama-cpp-localai-paged-cpu-all-build purge
|
||||
$(info $(GREEN)I llama-cpp-localai-paged build info:cpu-all-variants$(RESET))
|
||||
LLAMA_VERSION=$(LLAMA_VERSION) $(MAKE) -C $(CURRENT_MAKEFILE_DIR)/../llama-cpp-localai-paged-cpu-all-build llama.cpp
|
||||
$(call apply-paged-patches,$(CURRENT_MAKEFILE_DIR)/../llama-cpp-localai-paged-cpu-all-build/llama.cpp)
|
||||
SHARED_LIBS=ON EXTRA_CMAKE_ARGS="-DGGML_BACKEND_DL=ON -DGGML_CPU_ALL_VARIANTS=ON" TARGET="--target grpc-server --target ggml" LLAMA_VERSION=$(LLAMA_VERSION) \
|
||||
$(MAKE) -C $(CURRENT_MAKEFILE_DIR)/../llama-cpp-localai-paged-cpu-all-build grpc-server
|
||||
cp -rfv $(CURRENT_MAKEFILE_DIR)/../llama-cpp-localai-paged-cpu-all-build/grpc-server llama-cpp-localai-paged-cpu-all
|
||||
rm -rf ggml-shared-libs && mkdir -p ggml-shared-libs
|
||||
find $(CURRENT_MAKEFILE_DIR)/../llama-cpp-localai-paged-cpu-all-build/llama.cpp/build \( -name '*.so*' -o -name '*.dylib' \) -exec cp -av {} ggml-shared-libs/ \;
|
||||
@echo "Collected ggml shared backends:" && ls -la ggml-shared-libs/
|
||||
|
||||
llama-cpp-localai-paged-grpc:
|
||||
$(call paged-build,grpc,-DGGML_RPC=ON -DGGML_AVX=off -DGGML_AVX2=off -DGGML_AVX512=off -DGGML_FMA=off -DGGML_F16C=off -DGGML_BMI2=off,--target grpc-server --target ggml-rpc-server)
|
||||
|
||||
llama-cpp-localai-paged-rpc-server: llama-cpp-localai-paged-grpc
|
||||
cp -rf $(CURRENT_MAKEFILE_DIR)/../llama-cpp-localai-paged-grpc-build/llama.cpp/build/bin/ggml-rpc-server llama-cpp-localai-paged-rpc-server
|
||||
|
||||
package:
|
||||
bash package.sh
|
||||
|
||||
purge:
|
||||
rm -rf $(CURRENT_MAKEFILE_DIR)/../llama-cpp-localai-paged-*-build
|
||||
rm -rf llama-cpp-localai-paged-* package
|
||||
|
||||
clean: purge
|
||||
@@ -1,699 +0,0 @@
|
||||
# LocalAI paged-attention llama.cpp patch series
|
||||
|
||||
This backend vendors the patch series (in `patches/paged/`) that turns stock
|
||||
llama.cpp into LocalAI's paged-attention variant (`llama-cpp-localai-paged`). The
|
||||
patches are applied on top of a pinned upstream llama.cpp at build time; nothing
|
||||
here is a fork - it is a source-only `*.patch` stack plus this canonical doc.
|
||||
|
||||
> One-file rule: this README is the canonical reference for the patch series. The
|
||||
> only other docs are operational, kept in `docs/`, and linked below:
|
||||
> - [`PAGED_BITEXACT_NOTE.md`](docs/PAGED_BITEXACT_NOTE.md) - the per-path bit-exactness gate (the canonical paged-MoE md5 reference).
|
||||
> - [`LOCALAI_LLAMACPP_BACKEND_PLAN.md`](docs/LOCALAI_LLAMACPP_BACKEND_PLAN.md) - the design-of-record for shipping this as its own backend + the NVFP4 gallery items.
|
||||
> - [`VLLM_PARITY_FINAL.md`](docs/VLLM_PARITY_FINAL.md) - the definitive, closed record of the GB10 vLLM-parity investigation: full benchmark, every lever + verdict, the structural floors, and the parity verdict (summarized in section 9 below). Read this before reopening any parity work.
|
||||
> - [`EXECUTION_REARCH_SCOPE.md`](docs/EXECUTION_REARCH_SCOPE.md) - the reopened scope: ports vLLM's execution *architecture* (bf16-resident stream, expert-major fused MoE region, persistent-CTA GEMM, token-budget scheduler, blocked-solve GDN) into the fork additively, on the thesis that same-silicon 2-3x is software-architecture-conditional, not a hardware floor. Phased (P1-P6), each with a falsifiable P0 kill-gate. Read this to pick up parity work after `VLLM_PARITY_FINAL.md`.
|
||||
|
||||
---
|
||||
|
||||
## 1. What it is
|
||||
|
||||
`llama-cpp-localai-paged` is the LocalAI paged-attention llama.cpp backend: a
|
||||
vendored patch series over upstream llama.cpp that adds
|
||||
|
||||
- a **paged KV cache** (vLLM-style block manager: on-demand fixed-size blocks,
|
||||
free pool, ref-counted blocks) with a **block-table flash-attention** read so
|
||||
the attention kernels index physical cells instead of a contiguous buffer;
|
||||
- **cross-request prefix sharing** - concurrent requests that share a long
|
||||
prefix physically reuse one committed copy of the prefix blocks and prefill
|
||||
only their divergent suffix;
|
||||
- a **decode-first prefill scheduler** - a dynamic per-step prefill-token budget
|
||||
decoupled from `n_batch`, so a long prefill never freezes co-batched decode;
|
||||
- **GB10 / Blackwell NVFP4 decode optimizations** for the Qwen3.6 hybrid
|
||||
gated-DeltaNet (SSM) models, where the recurrent-state plumbing - not the FP4
|
||||
GEMM - dominates the decode step.
|
||||
|
||||
It is **pinned to llama.cpp `0ed235ea2c17a19fc8238668653946721ed136fd`** (kept == the stock `llama-cpp` backend's
|
||||
pin) and advanced only by a manual, bit-exact-gated pin-sync process (see
|
||||
section 7, "Pin + maintenance policy"), decoupled from the nightly auto-bumper. The pin must stay aligned with the stock pin because
|
||||
`grpc-server.cpp` is shared; an earlier bump to `c299a92c` was bit-exact but broke
|
||||
the grpc-server link and was reverted to the then-current stock pin.
|
||||
|
||||
The build gate is `LLAMA_PAGED` (default on in this tree); the paged engine is
|
||||
enabled per-model at runtime via the gallery `options:` knobs (`paged_kv:true`,
|
||||
`max_batch_tokens:`, `kv_unified:false`, ...). Against unpatched llama.cpp the
|
||||
runtime hooks are inert, so a single `grpc-server.cpp` is shared between the
|
||||
clean and the paged build.
|
||||
|
||||
---
|
||||
|
||||
## 2. Architecture
|
||||
|
||||
The decode step on these models breaks into three cost centers; the patch series
|
||||
attacks each one.
|
||||
|
||||
**Paged KV manager + block-table flash-attn.** A host-side `PagedKVManager`
|
||||
(`FreeBlockQueue` / `BlockPool` / chained-hash content cache) hands out
|
||||
fixed-size KV blocks on demand and reclaims them per-sequence (ref-counted, with
|
||||
copy-on-write for shared prefixes). The attention path reads through a **block
|
||||
table** - an `I32 [n_view, n_stream]` position-ordered physical-cell index passed
|
||||
as `src[5]` of `ggml_flash_attn_ext` - so the CUDA fattn vec/tile kernels and the
|
||||
CPU reference map logical KV index `j` to physical cell `block_table[seq*ne11+j]`
|
||||
and read K/V in place. Token-position ordering keeps the flash-attn online-softmax
|
||||
reduction order identical to stock. A null block table is the stock contiguous
|
||||
read, byte-identical.
|
||||
|
||||
**The gated-DeltaNet (GDN / SSM) decode path.** The Qwen3.6 hybrid models are 48
|
||||
gated-DeltaNet (linear-attention / SSM) layers + 16 full-attention layers. On
|
||||
GB10 the recurrent-state plumbing, not the weight GEMM, is the dominant decode
|
||||
cost. The series fuses that plumbing to mirror vLLM's
|
||||
`fused_recurrent_gated_delta_rule`: the recurrent state is read from and written
|
||||
to its cache slot in place (no copy-back, no `get_rows` materialization), the
|
||||
conv state is updated in place, the output projection is reshaped to route to the
|
||||
tensor-core MMQ GEMM, and the recurrence kernel is occupancy-retuned - all
|
||||
bit-exact (md5-gateable) against the f32 baseline.
|
||||
|
||||
**NVFP4 native FP4-MMA on Blackwell.** The NVFP4 dense/expert weight GEMM uses
|
||||
Blackwell's native FP4-MMA. The series removes a redundant activation-requantize
|
||||
in the MoE broadcast projections (bit-exact byte copy of identical blocks) and
|
||||
keeps CUDA graphs on for the grouped-MMQ MoE decode step. These are the only
|
||||
NVFP4-specific optimizations; on non-Blackwell hardware the FP4 path falls back
|
||||
to dequant.
|
||||
|
||||
**The prefill/decode scheduler.** `update_slots()` already emits one unified
|
||||
mixed prefill+decode batch per step. The scheduler patches change only the *count*
|
||||
of prefill tokens admitted per step: decode tokens are claimed first
|
||||
(decode-first), then a dynamic budget `max(n_ubatch, T - D)` (where `D` is the
|
||||
live decode load and `T` is `LLAMA_MAX_BATCH_TOKENS`) admits prefill, auto-
|
||||
shrinking as decode load rises. Pure scheduler policy, byte-identical when off,
|
||||
orthogonal to the paged allocator.
|
||||
|
||||
---
|
||||
|
||||
## 3. Patch series (0001-0063)
|
||||
|
||||
Source-only patches, with intentional numbering gaps (e.g. 0005, 0027). The
|
||||
decode-serving graph-reuse levers are 0040-0041. "Bit-exact" = greedy md5 /
|
||||
`test-backend-ops` byte-identical to the relevant baseline; the gate methodology
|
||||
is in section 5.
|
||||
|
||||
### Paged-KV core (0001-0012)
|
||||
|
||||
| # | What it does | Bit-exact |
|
||||
|---|---|---|
|
||||
| 0001 | Vendor the host-side paged KV block manager (`FreeBlockQueue`, `BlockPool`, `PagedKVManager`, chained-hash prefix cache). Pure C++17, nothing uses it yet. | n/a (no behavior) |
|
||||
| 0002 | Place each sequence at permuted, non-contiguous block positions in `find_slot` (proves attention is invariant to physical KV placement). | yes (token-identical) |
|
||||
| 0003 | Gather K/V/mask down to each stream's non-empty cells before `build_attn_mha`, position-sorted so the FA reduction order matches stock. | yes |
|
||||
| 0004 | Drive paged placement through the vendored manager: blocks popped on demand, returned on seq end. Core kv-cache struct untouched. | yes (stock path byte-identical) |
|
||||
| 0006 | Host-side cross-request prefix caching: hash prefix blocks, reuse matching physical blocks (ref-count++), COW-privatise before a divergent write. | yes (default off) |
|
||||
| 0007 | Wire the prefix cache into the engine so a new sequence physically shares cached prefix blocks and skips recomputing the shared prefix. | yes (verified byte-identical) |
|
||||
| 0008 | Wire cross-request prefix share into the llama-server continuous-batch loop so concurrent shared-prefix requests prefill only the suffix (36x fewer prefill tokens at K=32). | within CUDA batch-shape non-determinism band |
|
||||
| 0009 | Replace the per-step gather with an **in-kernel paged read** (block table as `src[5]`); the K/V `get_rows` is gone. Decode step at batch32 691->696ms (was 1279ms gathered). | yes on CPU/batch1; GPU batch>1 within vec-vs-mma band |
|
||||
| 0010 | Graft the block-table read into the tile kernel; add a dispatch guard so a present block table routes ONLY to vec/tile (never the mma/wmma kernels that ignore it). | yes (CPU byte-identical; vec route) |
|
||||
| 0011 | Route the GQA-grouped F16 decode to the **tile kernel** (native head-group reuse) by default; vec for everything else. Paged decode to within 1.8% of stock. | vs stock-mma: different-kernel rounding; bit-exact vs vec |
|
||||
| 0012 | Defensive `GGML_ASSERT(n_view % 64 == 0)` so a future pad/tile change can't silently reintroduce a past-end KV leak on the tile route. | yes (additive assert) |
|
||||
|
||||
### Decode-first scheduler (0013, 0016)
|
||||
|
||||
| # | What it does | Bit-exact |
|
||||
|---|---|---|
|
||||
| 0013 | `LLAMA_PREFILL_BUDGET`: a static per-step prefill-token budget decoupled from `n_batch` (vLLM `--max-num-batched-tokens` analogue). Flattens the decode ITL spike a long prefill inflicts (8.5x smaller worst freeze). | yes (off/short = byte-identical; == `-b` chunking) |
|
||||
| 0016 | Supersede 0013 with a **dynamic decode-first** budget: `max(n_ubatch, T-D)`, auto-shrinking as decode load `D` rises. Policy-only inside `update_slots()`, zero libllama changes. | yes (default-off byte-identical) |
|
||||
|
||||
(0014/0015 are the MoE token-tile levers: 0014 adds `LLAMA_MOE_MMQ_X` (opt-in
|
||||
high-batch decode micro-opt, +4.8% on Qwen3-Coder-30B), 0015 makes it a
|
||||
default-on, density-aware auto-select that is prefill-safe by construction. Both
|
||||
bit-exact. 0017 is the dense FP4-GEMM occupancy-tune track: bit-exact gate green,
|
||||
but every cheap occupancy lever regressed on GB10, so nothing is enabled - it
|
||||
ships as the parity gate + default-off instrumentation only.)
|
||||
|
||||
### Decode-serving graph reuse (0040, 0041)
|
||||
|
||||
These two close the **continuous-serving** decode gap (distinct from the static
|
||||
batched-bench decode kernel, which is already at vLLM parity - see
|
||||
[`docs/DECODE_SERVING_SCOPE.md`](docs/DECODE_SERVING_SCOPE.md)). In serving the
|
||||
host rebuilt the ggml graph on **every** decode step (layer-A graph reuse was 0%),
|
||||
so the GPU idled while the host rebuilt - the host-bound -39% the static bench
|
||||
hides.
|
||||
|
||||
| # | What it does | Bit-exact |
|
||||
|---|---|---|
|
||||
| 0040 | **S1 paged decode-graph reuse** - the paged decode inputs (`input_block_table` / `input_gather_idxs`) never overrode `can_reuse` (defaults to false), so any graph carrying a paged input could never be reused. Add a correct `can_reuse` keyed on the (256-bucketed) block-table dims + a live-mctx refresh from the owning attn input. `LLAMA_PAGED_NO_GRAPH_REUSE=1` forces the pre-S1 path. | yes (md5 byte-identical reuse on/off; dense `5951a5b4`, paged-MoE `8cb0ce23`) |
|
||||
| 0041 | **S3 decode-shape-stable scheduling** - keep co-batched prefill OUT of decode steps so the pure-decode batch shape stays reuse-stable (S1 makes a pure-decode step reusable; S3 makes the scheduler emit them). Pure `update_slots()` policy on top of 0016; prefill admitted on a bounded cadence (`LLAMA_PAGED_PREFILL_PERIOD`, default 8). **Default OFF** (opt-in via `LLAMA_PAGED_DECODE_STABLE=1`): a measured end-to-end A/B proved default-on is a serving mistake - deferring prefill admission on the period-8 cadence gives **2.5x worse TTFT** (60s vs 24s at N=256) and **20-29% lower end-to-end throughput**, with no end-to-end win at any concurrency; its apparent `decode_agg` gain was a metric artifact (faster per-step decode bought by starving prefill). Default prefers prompt prefill admission for good TTFT; opt in only for decode-dominated, low-arrival traffic where TTFT is not a concern. | yes (byte-identical on/off; per-stream independent in serving) |
|
||||
|
||||
Measured (GB10, MoE Qwen3.6-35B-A3B-NVFP4, 128-client staggered streaming load):
|
||||
graph reuse **0% -> 72.2%**, host window `hostproc` **15.98 -> 6.31 ms/step**,
|
||||
decode **4.05 -> 5.52 tok/s/seq median (4.24 -> 5.96 mean, at vLLM's ~5.9
|
||||
sustained)**. S1 is necessary but **not** sufficient alone (13.8% reuse - prefill
|
||||
co-batching churns the shape nearly every step); S3 is the multiplier of that
|
||||
per-step decode metric. **But those are per-step decode numbers, not an end-to-end
|
||||
serving win**: a later end-to-end A/B showed S3-default-on regresses real serving
|
||||
(2.5x worse TTFT, 20-29% lower end-to-end throughput, no win at any concurrency),
|
||||
because the period-8 cadence defers prefill admission. So **only S1 (0040) ships
|
||||
default-on; S3 (0041) now defaults OFF and is opt-in** (`LLAMA_PAGED_DECODE_STABLE=1`,
|
||||
for decode-dominated low-arrival traffic). The static batched-bench A/B isolates the S1
|
||||
mechanism: paged decode reuse 0% -> 95.5% (throughput flat there, since the static
|
||||
regime is GPU-bound). **S2 (double-buffer `set_inputs`) was dropped**: the Phase-0
|
||||
profile put `set_inputs` at ~0.05 ms/step (the cost is the rebuild, not the input
|
||||
copy), so it has nothing to recover. The remaining ~28% serving rebuilds are
|
||||
request-boundary D/seq-set churn + the prefill-cadence steps. A **padded/fixed-slot
|
||||
decode shape** to capture them was then implemented and GPU-tested (2026-06-28) and
|
||||
**REJECTED** - it is bit-exact/inert but regresses serving throughput at every
|
||||
concurrency, because this serving decode is GPU-compute-bound (baseline reuse 0% ~=
|
||||
S1+S3 reuse 72% on aggregate tok/s), so the dummy-row compute it adds costs more
|
||||
than the reuse it recovers. Full record + numbers in `docs/DECODE_SERVING_SCOPE.md`
|
||||
("Padded-shape lever - rejected").
|
||||
|
||||
### Prefill fusions (0042, 0044)
|
||||
|
||||
CUDA-family graph fusions of the pre-norm residual chain and the gated-DeltaNet
|
||||
output norm: separate `rms_norm` / `mul` / `add` / `silu` launches collapse into
|
||||
one kernel so the intermediate never round-trips to HBM. Bit-exact (the fused
|
||||
kernel reproduces the unfused FP order; float multiply is commutative). Each is
|
||||
env-gated default-ON (`LLAMA_FUSE_*=0` for a clean single-build A/B that reverts
|
||||
to the byte- and kernel-identical unfused path).
|
||||
|
||||
| # | What it does | Bit-exact / effect |
|
||||
|---|---|---|
|
||||
| 0042 | **Fused residual-add + RMS norm + weight multiply** (`rms_norm_pre_add_mul_f32`) - the pre-norm residual `h = x + sub_out; n = rms_norm(h) * w` ran as a `k_bin_bcast` ADD feeding the fused rms_norm+mul; the residual ADD has a second consumer (the skip add) so it can't pass the single-use `ggml_can_fuse`. Recognized via `ggml_can_fuse_subgraph` (ADD + final MUL both outputs), folded into one launch that publishes `h` and emits `scale * h * w`. Gate `LLAMA_FUSE_ADD_RMSNORM`. | yes (dense `5951a5b4`, MoE `8cb0ce23`); dense S_PP +0.5% |
|
||||
| 0044 | **Fused gated RMSNorm + SiLU gate multiply** (`rms_norm_gate_mul_f32`) - the gated-DeltaNet output norm `(rms_norm(x) * w) * silu(z)` (qwen35 / qwen35moe `build_norm_gated`) ran as rms_norm_mul + silu_mul, two launches with the normalized intermediate crossing HBM. The gate z-projection (a MUL_MAT) is scheduled between the weight MUL and the SILU, so the chain is not naturally consecutive; `build_norm_gated` emits the gate multiply as `mul(silu(z), normalized)` (commutative, bit-exact) so the graph lays out the consecutive subgraph `{ SILU, RMS_NORM, MUL, MUL }` that `ggml_cuda_can_fuse` folds into one `scale * x * w * silu(z)` launch. Gate `LLAMA_FUSE_GATE_RMSNORM`. Profile (dense npp512): 672 (rms_norm_mul + silu_mul) -> 336 fused launches. | yes (dense `5951a5b4`, MoE `8cb0ce23`, paged + non-paged; `test-backend-ops` 12979/12979); S_PP dense +1.1% (~+10 us/tok), MoE +0.9% |
|
||||
|
||||
### SSM (gated-DeltaNet) decode levers (0018-0022, 0028)
|
||||
|
||||
These are the dominant decode levers on the Qwen3.6 hybrid models. All bit-exact.
|
||||
|
||||
| # | What it does | Effect (dense q36-27b / MoE q36-35b-a3b @npl128) |
|
||||
|---|---|---|
|
||||
| 0018 | **In-place SSM state write-back** - the recurrence writes its final state directly into the cache slot, removing the ~225MB/copy D2D memcpy (18.9% of decode time). | dense +23.5% / MoE +18.9% |
|
||||
| 0019 | **Fused recurrent-state gather** - the op reads each sequence's prior state directly from `cache[ids[seq]]` (no `get_rows` materialization); race-free in-place + ids read. | dense +37.8% / MoE +35.3% |
|
||||
| 0020 | **o_proj MMVQ->MMQ reshape** - collapse the GDN output to 2D so the output projection routes to the M=128 tensor-core MMQ GEMM (was a batch<=8 MMVQ GEMV). The single biggest decode-parity lever. | dense +31.7% (->85.9% of vLLM) / MoE +23.3% |
|
||||
| 0021 | **Conv-state in-place fusion** - one `ggml_ssm_conv_update_inplace` op replaces the 4-op conv chain (transpose+concat+conv+silu+ring-cpy), writing the shifted ring state in place. | dense +3.2% / MoE +3.5% |
|
||||
| 0022 | **GDN recurrence occupancy/coalescing retune** - column-folding (NUM_WARPS/COLS_PER_WARP) raises memory-level parallelism on the bandwidth-bound B=128 recurrence kernel; per-column f32 FMA order unchanged. 73.4%->84.6% of GB10 peak BW. | dense +11.1% / MoE +8.3% |
|
||||
| 0028 | **Recurrent conv-tap gather fusion** - the last `k_get_rows` in the GDN decode path (the conv-state tap gather) becomes an indexed in-kernel read. | dense ~377 t/s / MoE ~784 t/s |
|
||||
|
||||
### MoE NVFP4 quant (0023, 0025, 0043)
|
||||
|
||||
| # | What it does | Bit-exact |
|
||||
|---|---|---|
|
||||
| 0023 | **NVFP4 activation-quantize de-dup** - the broadcast up/gate projections re-quantize the same token activation once per expert; quantize the unique token activations once and byte-copy them into the expert-gathered layout. The only NVFP4-specific patch. | yes (byte-identical) |
|
||||
| 0025 | **MoE decode re-graph** - keep CUDA graphs on for the grouped-MMQ MoE decode step (the upstream guard disables graphs conservatively; the grouped path has no host sync). Was env-gated `LLAMA_MOE_FORCE_GRAPHS`; now ON by default via 0043. | yes (graph replay re-issues identical kernels) |
|
||||
| 0043 | **MoE decode graph default-on (D1)** - flip 0025 to ON by default: capture/replay the full-step decode CUDA graph (incl. the grouped-MMQ MoE dispatch) instead of re-issuing every kernel each step. Guard is `should_use_mmq()` (FALSE for the large-M NVFP4 prefill of 0034, so prefill keeps graphs disabled - its per-expert host-loop genuinely syncs). `LLAMA_MOE_NO_FORCE_GRAPHS=1` forces the conservative pre-0025 disable for A/B. D1 profiling: the per-expert host-loop (the only device->host MoE-routing readback) is never hit on the NVFP4 grouped path (sync count identical graphs on/off); steady decode is ~99% GPU-busy, so the cost removed is per-step host kernel RE-ISSUE, not a sync. | yes (md5 byte-identical default/off/forced; paged-MoE `8cb0ce23`, dense `5951a5b4`) |
|
||||
|
||||
### Pool reclaim, block-table cache, backend gate
|
||||
|
||||
| # | What it does | Bit-exact |
|
||||
|---|---|---|
|
||||
| 0024 | **Paged-pool burst-reclaim** - truncate trailing blocks on partial-tail `seq_rm`, defrag the free queue when idle, release blocks on slot completion. Fixes the long-server burst-degradation bug (post-burst prefill collapse 488->44 t/s, restored to 532). Host-side accounting only. | yes |
|
||||
| 0029 | **Block-table within-step host cache** - the block table is fixed for the whole step; cache it on first build and memcpy it for the other full-attention layers (get_block_table -87%/-91%). | yes, per path (paged-MoE ref `8cb0ce23`) |
|
||||
| 0030 | **Fused-op backend gate** - the fused GDN / discriminated SSM_CONV ops are CUDA-family + CPU only; force them off on any non-CUDA compute backend so a Vulkan/SYCL/Metal build can't silently run the wrong plain-conv kernel. | yes on CUDA (byte-identical pre-0030); safety gate elsewhere |
|
||||
| 0031 | **Chunked parallel-scan GDN prefill kernel** (upstream TODO) - FLA-style chunked gated-delta-rule for prefill (non-KDA / f32 / final-state): intra-chunk delta rule solved in parallel (UT-transform + forward subst), inter-chunk recurrence over n_tokens/C steps. The scalar-serial form (`GDN_TC=0`) was bit-exact-benign but not faster than the tuned sequential scan at the GB10-forced C=16 (see section 5); **superseded for paged by the tensor-core M5 path of 0047**. | NEW per-path (`test-backend-ops` 91/91, <=1e-7 NMSE vs CPU ref) |
|
||||
| 0047 | **GDN M5 tensor-core chunked-scan prefill, f32-only re-port, default-ON under paged KV** - the f32/tf32 tensor-core forms of 0031's scan (KK/QK Gram = M2, KS/QS state-boundary 3xtf32 = M3, P*U output = M4, full form-T solve + state-update mma = M5), single build, runtime-selected by `GDN_TC`. Ships **M5 default-on when `LLAMA_KV_PAGED` is set** (`GDN_TC=5` + `GDN_CHUNK_MIN=64`, both env-overridable; OFF/`INT_MAX` when not paged). `GDN_CHUNK_MIN` is the per-call engage threshold and stays > 1 so decode (1 tok/call) keeps the sequential recurrence (at 1 it swallows decode and drops S_TG ~25%); 64 tuned from a {1,32,64,128,256} sweep. The bf16/hybrid dev-tree machinery (STATE_BF16/HYBRID, the dropped 0026 ssm_bf16_tau) and the bf16 CONFIG-C (M8) plus register-resident M6/M7 variants are NOT part of this f32-only series. MoE prefill S_PP +3.5% @npp512 (3x A/B), +17.7% @npp2048; decode S_TG unchanged. | NEW per-path, benign (`test-backend-ops` GATED_DELTA_NET 46/46 default AND force-M5, incl. multi-chunk/tail-chunk/multi-seq; greedy md5 default-on == M5-forced == canonical on the gate prompt: paged-MoE `8cb0ce23`, dense `5951a5b4`; long MoE prompt = one benign greedy flip vs sequential, dense byte-identical) |
|
||||
| 0046 | **GDN prefill geometry gated by scan length** - patch 0022's `(NUM_WARPS=16, COLS_PER_WARP=8)` column-fold of the GDN sequential-recurrence dispatch (`case 128`) is a decode win but was applied UNCONDITIONALLY, so it also hit dense prefill (~-6% vs stock): on a long sequential scan the launch `grid.z` collapses from `S_v/4 = 32` to `S_v/(16*8) = 1` and the SMs starve (profiled: `gated_delta_net` +54% GPU time = the whole dense-prefill regression). Gate the geometry by per-call scan length: long scans (prefill, `n_tokens >= GDN_PREFILL_NTOK`, default 256) take stock's high-grid.z `(4,1)` geometry; short scans (decode) keep the `(16,8)` retune. Recovers dense prefill +7.2% back to stock parity, keeps the decode win. `GDN_PREFILL_NTOK` tunes the crossover; an explicit `GDN_NW`/`GDN_CPW` sweep still overrides (gate yields when either is set), so the one-build %peak A/B harness is unchanged. | yes (patch 0022 proved every `{NW,CPW}` variant byte-identical, so switching geometry by scan length cannot move the md5) |
|
||||
|
||||
### Speculative / MTP investigation (0054, 0055)
|
||||
|
||||
| # | What it does | Bit-exact / effect |
|
||||
|---|---|---|
|
||||
| 0054 | **Disable backend sampling for MTP drafts** - forces server MTP draft generation through the target-side sampler acceptance path instead of letting the draft backend sample independently. This was required for the Phase 14 rollback/prefix safety gate. | yes for canonical non-MTP gates; Phase 14 MTP normalized greedy-prefix gate passed |
|
||||
| 0055 | **Trace speculative batch shapes** - adds default-off `LLAMA_SPEC_SHAPE_TRACE=1` server logs around `server_slot::handle_last_sampled_token()`, reporting normal decode rows and MTP verification `K + 1` rows (`draft`, `outputs`, `spec_i_first`, `spec_i_last`). This is instrumentation only for Phase 18 shape-entropy measurement before any scheduler experiment. | yes (env unset is silent; DGX gates after patch: MoE `8cb0ce23`, dense `5951a5b4`, `MUL_MAT_ID` `806/806`) |
|
||||
| 0056 | **Trace MoE MMQ batch shapes** - adds default-off `LLAMA_MOE_MMQ_SHAPE_TRACE=<n>` logs from the grouped-MMQ host selector, reporting routed assignment count, estimated active experts, density, selected `mmq_x`, `mmq_y`, and stream-k. This is evidence-only instrumentation for sizing structural grouped-MMQ work after Phase 28 rejected launch-bounds/row-tile knobs. | yes (env unset and trace-enabled gates both green: MoE `8cb0ce23`, dense `5951a5b4`, `MUL_MAT_ID` `806/806`; trace cap verified with 4 lines) |
|
||||
| 0057 | **Trace MoE MMQ launch shapes** - extends `LLAMA_MOE_MMQ_SHAPE_TRACE=<n>` with bounded `[LLAMA_MOE_MMQ_LAUNCH]` lines from `launch_mul_mat_q`, recording actual `ntiles_dst`, `stream_k_blocks`, tile efficiency, `fixup`, `ntx/nty/ntzw`, and compiled `mmq_x/mmq_y`. This is evidence-only instrumentation to distinguish real stream-k/fixup overhead from small-M kernel-shape cost. | yes (default-off, trace-enabled, and post-serving gates green: MoE `8cb0ce23`, dense `5951a5b4`, `MUL_MAT_ID` `806/806`; Phase 31 n128 trace showed decode and prefill `fixup=0`, `stream_k_blocks == ntiles_dst`) |
|
||||
| 0058 | **Trace MoE small-M MMQ candidates** - adds `LLAMA_MOE_MMQ_SMALL_M_TRACE=<n>` and a host-only classifier for decode-like low-density grouped-MMQ shapes (`ncols_max <= 128`, density `<=4`, `mmq_x_best <=64`). It only counts candidate calls for the next structural tile-policy A/B; no numeric branch is added. | yes (default-off, trace-enabled, and post-serving gates green: MoE `8cb0ce23`, dense `5951a5b4`, `MUL_MAT_ID` `806/806`; Phase 32 n128 trace found 4096 candidates, mostly `mmq_x_best=64/48`) |
|
||||
| 0059 | **Gate MoE small-M MMQ tile policy** - adds default-off `LLAMA_MOE_SMALL_M_TILE=<n>` to cap only classified small-M MoE grouped-MMQ calls. This was used to A/B vLLM-like smaller M blocks without changing default inference. | yes (default-off, tile16, tile8, and post-serving gates green: MoE `8cb0ce23`, dense `5951a5b4`, `MUL_MAT_ID` `806/806`; Phase 33 rejected tile16 and tile8 as slower) |
|
||||
| 0060 | **Trace MoE MMID dispatch routes** - adds default-off `LLAMA_MOE_MMID_ROUTE_TRACE=<n>` around `MUL_MAT_ID` dispatch, classifying each call as `mmvq`, `mmvf`, grouped `mmq`, `mmf`, or host-sync `fallback`. This is evidence-only instrumentation to resolve whether serving hits the per-expert host-sync fallback. | yes (default-off, trace-enabled, and post-serving gates green: MoE `8cb0ce23`, dense `5951a5b4`, `MUL_MAT_ID` `806/806`; Phase 34 n128 trace found `mmq=2776`, `mmvq=1320`, `host_sync=0/4096`) |
|
||||
| 0061 | **Trace regular MUL_MAT dispatch routes** - adds default-off `LLAMA_MUL_MAT_ROUTE_TRACE=<n>` around regular `MUL_MAT`, classifying projection-heavy calls as `vec_f`, `mat_f`, `vec_q`, `mmq`, `batched_cublas`, `op_*`, `fp4_prefill`, or `fwht`. This is evidence-only instrumentation for the `bf16-proj` serving bucket. | yes (default-off, trace-enabled, and post-serving gates green: MoE `8cb0ce23`, dense `5951a5b4`, `MUL_MAT` `1146/1146`, `MUL_MAT_ID` `806/806`; Phase 35 n128 trace found BF16 routes `mat_f=2485`, `op_cublas=1330`) |
|
||||
| 0062 | **Trace cuBLAS subroutes** - adds default-off `LLAMA_CUBLAS_ROUTE_TRACE=<n>` around the generic cuBLAS `MUL_MAT` path, classifying calls as `nvfp4_bf16_tc`, `bf16_tc`, `f16_tc_32f`, `f16_tc_16f`, or `sgemm`. This is evidence-only instrumentation for the Phase 35 `op_cublas` bucket. | yes (default-off, trace-enabled, and post-serving gates green: MoE `8cb0ce23`, dense `5951a5b4`, `MUL_MAT` `1146/1146`, `MUL_MAT_ID` `806/806`; Phase 36 n128 trace found `bf16_tc=5681`, `sgemm=2511`) |
|
||||
| 0063 | **Trace cuBLAS tensor names** - extends `LLAMA_CUBLAS_ROUTE_TRACE=<n>` with `src0`, `src1`, and `dst` names so the `sgemm` bucket can be tied back to graph nodes. | yes (default-off, trace-enabled, and post-serving gates green: MoE `8cb0ce23`, dense `5951a5b4`, `MUL_MAT` `1146/1146`, `MUL_MAT_ID` `806/806`; Phase 37 n128 trace identified `sgemm` as `ffn_gate_inp* -> ffn_moe_logits/shared_expert_gate`) |
|
||||
|
||||
> **Dropped: patch 0026 (hybrid per-head bf16 SSM state, `ssm_bf16_tau`).** Once
|
||||
> the decode fusions (0028 recurrent-state gather-fusion + 0029 block-table cache)
|
||||
> landed, the bf16-SSM lever bought nothing: a clean re-measurement forcing **all**
|
||||
> gated-DeltaNet heads to bf16 (`tau=100000`) gives **flat** decode (780.6 vs
|
||||
> 780.0 t/s) - the mode engages but adds zero throughput because it is subsumed by
|
||||
> the fusions. It was a precision trade (not bit-exact) plus extra bug surface and
|
||||
> CUDA template-instantiation compile cost with no benefit, so it was removed. See
|
||||
> section 5 ("rejected / flat levers") for the full record.
|
||||
|
||||
---
|
||||
|
||||
## 4. Benchmarks
|
||||
|
||||
Hardware: **GB10 / DGX Spark** (CUDA 13, sm_121). Models: dense
|
||||
**Qwen3.6-27B-NVFP4** and MoE **Qwen3.6-35B-A3B-NVFP4**. Metric: `decode_agg`
|
||||
S_TG (t/s) from `llama-batched-bench`, `-fa on -ngl 99`, `npp 128 / ntg 128`,
|
||||
swept over serving width `npl` in {8, 32, 64, 128}. Plots:
|
||||
[`qwen36_decode_overview.png`](docs/qwen36_decode_overview.png) (both models),
|
||||
[`qwen36_dense_decode_vs_npl.png`](docs/qwen36_dense_decode_vs_npl.png),
|
||||
[`qwen36_moe_decode_vs_npl.png`](docs/qwen36_moe_decode_vs_npl.png); raw data
|
||||
[`final_benchmark.csv`](docs/final_benchmark.csv).
|
||||
|
||||

|
||||
|
||||
> The plot above also shows a third "bf16-tau" llama curve. That was the opt-in
|
||||
> `ssm_bf16_tau` lever (patch 0026), since **dropped** - a clean re-measurement
|
||||
> showed it flat once the decode fusions landed (see section 5). The numbers below
|
||||
> use only **stock** vs **patched** vs **vLLM**.
|
||||
|
||||
> **What was re-measured (2026-06-27).** The two llama columns - **stock** and
|
||||
> **patched** - were re-measured this session on one consistent
|
||||
> `llama-batched-bench` harness. The **vLLM** column is the **prior-session
|
||||
> reference** (kept as-is, *not* re-run this session). Per-run peak
|
||||
> VRAM was *not* re-captured: the GB10's unified Grace-Blackwell LPDDR5x reports
|
||||
> `[N/A]` to `nvidia-smi --query-gpu=memory.used` and the bench does not print it
|
||||
> (the memory-advantage note below is the prior-session finding).
|
||||
|
||||
### (a) + (b) Patched vs stock vs vLLM
|
||||
|
||||
The **stock** column is a separate, unpatched llama.cpp built at this backend's
|
||||
**exact pin (`9d5d882d`)**; the **patched** column is
|
||||
the paged binary, env/flag-toggled (`LLAMA_KV_PAGED=1`, plus
|
||||
`LLAMA_MOE_FORCE_GRAPHS=1` for MoE). Both
|
||||
run on the **same harness**, so "x over stock" is an apples-to-apples measure of
|
||||
the patch series. (Note: the patch series' dominant SSM decode fusions are
|
||||
compiled in, not env-gated - toggling `LLAMA_KV_PAGED` alone on the *patched*
|
||||
binary does **not** reproduce stock; only the separately-built unpatched
|
||||
`9d5d882d` binary does.) The **vLLM** column is a **different harness** (vLLM
|
||||
server + client continuous batching) and a **prior-session reference**, so the
|
||||
cross-engine "% of vLLM" is **indicative, not apples-to-apples**.
|
||||
|
||||
**Dense Qwen3.6-27B-NVFP4** (decode t/s):
|
||||
|
||||
| npl | stock | patched | vLLM (prior) | patched x over stock |
|
||||
|----:|------:|--------:|-------------:|---------------------:|
|
||||
| 8 | 68.3 | 85.3 | 70.4 | 1.25x |
|
||||
| 32 | 119.9 | 211.9 | 211.8 | 1.77x |
|
||||
| 64 | 142.8 | 305.2 | 309.1 | 2.14x |
|
||||
| 128 | 155.1 | 382.1 | 418.8 | 2.46x |
|
||||
|
||||
Dense **patched** is parity-to-ahead of vLLM (121 / 100 / 99 / 91% of vLLM across
|
||||
the widths).
|
||||
|
||||
**MoE Qwen3.6-35B-A3B-NVFP4** (decode t/s):
|
||||
|
||||
| npl | stock | patched | vLLM (prior) | patched x over stock |
|
||||
|----:|------:|--------:|-------------:|---------------------:|
|
||||
| 8 | 186.7 | 230.3 | 256.5 | 1.23x |
|
||||
| 32 | 267.4 | 466.4 | 500.8 | 1.74x |
|
||||
| 64 | 320.5 | 622.4 | 686.1 | 1.94x |
|
||||
| 128 | 347.2 | 784.3 | 882.2 | 2.26x |
|
||||
|
||||
MoE **patched** is 90 / 93 / 91 / 89% of vLLM.
|
||||
|
||||
**Caveat on the vLLM column.** It is a **different harness** and a
|
||||
**prior-session** measurement (not re-run this session), so the cross-engine "% of
|
||||
vLLM" is **indicative, not apples-to-apples**. Memory (prior session): llama uses
|
||||
**1.5-3x lower** memory than vLLM.
|
||||
|
||||
**Takeaway.** Re-measured this session, the patch series gives up to **2.46x
|
||||
(dense) / 2.26x (MoE)** over true-stock `9d5d882d` on the same harness (close to,
|
||||
slightly below, the prior 2.59x / 2.33x - llama was re-measured, vLLM kept).
|
||||
Dense is parity-to-ahead of vLLM; MoE **patched** sits at ~89-93% of the
|
||||
prior-session vLLM. The residual MoE gap is structural (see section 5).
|
||||
|
||||
### (c) Apple Silicon (M4, 16GB Metal) - does the patchset help here?
|
||||
|
||||
Short answer: **no - the wins are CUDA/Blackwell-specific.** Two facts first: the
|
||||
24GB NVFP4 GGUF doesn't fit a 16GB M4 (SSD paging), and on Metal `supports_op`
|
||||
**excludes NVFP4** from `MUL_MAT`/`MUL_MAT_ID`/`GET_ROWS` (FP4 matmuls fall back to
|
||||
CPU - no Apple FP4-MMA). So NVFP4 Qwen3.6 is not a Mac fit; a Metal-native Q4_K is.
|
||||
|
||||
Measured **stock vs patched** (same pin `c299a92c`, both built `-DGGML_METAL=ON`;
|
||||
the 28-patch series **compiles clean on Metal** - the CUDA code is `#if`-guarded),
|
||||
on **Qwen3-8B Q4_K_M** (a dense GQA model that fits 16GB and exercises the *live*
|
||||
Metal features; no Qwen3.6 hybrid GGUF fits 16GB, and the GDN fusions gate off on
|
||||
Metal anyway), `llama-bench` pp512/tg128 t/s:
|
||||
|
||||
| config | pp512 | tg128 |
|
||||
|---|---:|---:|
|
||||
| stock | 226.7 | 20.4 |
|
||||
| patched, paged **off** | 226.7 | 20.3 (= stock) |
|
||||
| patched, paged **on** | 222.6 | 19.8 (~0.97x) |
|
||||
|
||||
Concurrency (`batched-bench`) scales identically to stock (S_TG ~20 -> ~137 at
|
||||
npl32, from llama.cpp's existing batching). **Verdict: neutral-to-slightly-negative
|
||||
on Metal.** Patched-paged-off equals stock; turning paged on is ~0-3% slower
|
||||
decode / ~2-8% slower prefill, because the in-kernel block-table flash-attn read
|
||||
that *recovers* the gather cost is CUDA-only (`fattn-*.cuh`) - on Metal the paged
|
||||
path falls back to a host-side gather, pure overhead over stock's contiguous read.
|
||||
Everything Blackwell-specific (NVFP4, GDN fusions via 0030, occupancy) is inert.
|
||||
So **on Apple Silicon, prefer the stock `llama-cpp` backend.**
|
||||
|
||||
**Vulkan / SYCL** (source analysis): the gated-DeltaNet and SSM_CONV ops DO have
|
||||
upstream kernels on Vulkan and SYCL (as on Metal), so the Qwen3.6 hybrids RUN on
|
||||
all three via the non-fused path. The patchset's fusions are gated off there
|
||||
(0030), so the outcome is the same neutral-to-slightly-negative as Metal - not
|
||||
"won't run". This backend therefore ships **CUDA-only** (where the fusions are
|
||||
live + verified); non-CUDA users should use the stock `llama-cpp` backend. See
|
||||
[`UPSTREAM_LAYER2_SCOPE.md`](docs/UPSTREAM_LAYER2_SCOPE.md) for what native non-CUDA
|
||||
fused kernels would take.
|
||||
|
||||
---
|
||||
|
||||
## 5. Dev notes - what we learned
|
||||
|
||||
**Bit-exact methodology.** Every bit-exact patch is gated two ways: (1) a greedy
|
||||
md5 gate - `llama-completion -m MODEL -ngl 99 -fa on -p "The capital of France
|
||||
is" -n 48 --temp 0 --seed 1 | md5sum`, paged paths prefixed with
|
||||
`LLAMA_KV_PAGED=1` (+ `LLAMA_MOE_FORCE_GRAPHS=1` for paged MoE), on the default
|
||||
chat-template path; and (2) `test-backend-ops` (CUDA0 vs CPU oracle) for every
|
||||
touched op (`SSM_CONV*`, `GATED_DELTA_NET`, `MUL_MAT`, `MUL_MAT_ID`).
|
||||
For DGX work, `paged-inference-gates.sh` runs the canonical MoE/dense transcript
|
||||
md5 checks and selected `test-backend-ops` filters, and refuses to start while
|
||||
docker, `local-ai-worker`, GPU compute processes, or a non-free GPU lock are
|
||||
present.
|
||||
|
||||
For direct `llama-server` MTP serving A/B work, use
|
||||
`paged-mtp-serving-bench.sh`. It runs the same pre/post inference gates, compares
|
||||
baseline vs `--spec-type draft-mtp`, and captures the h2h client summaries plus
|
||||
MTP acceptance lines. Phase 15 rejected current MTP serving on GB10 despite
|
||||
passing safety gates; do not enable it by default.
|
||||
|
||||
**The gate is per-path** (see [`PAGED_BITEXACT_NOTE.md`](docs/PAGED_BITEXACT_NOTE.md)).
|
||||
Dense is bit-exact across paged/non-paged (`5951a5b4`). The **paged MoE** md5
|
||||
(`8cb0ce23`) does **not** byte-match the **non-paged MoE** md5 (`07db32c2`); this
|
||||
is a benign FP-accumulation-order difference of the paged attention reduction,
|
||||
**KL-validated** against the f16 reference: KLD(paged||f16) 0.13600 <=
|
||||
KLD(nonpaged||f16) 0.13660, PPL within +/-0.29, ~zero probability bias - two
|
||||
equivalent FP-reorderings of the same quantized model, not a regression. Future
|
||||
paged-MoE regressions therefore compare to `8cb0ce23`, not `07db32c2`.
|
||||
|
||||
**MoE-parity conclusion** (the residual gap is structural). The two heaviest MoE
|
||||
decode kernels - the GDN-SSM recurrence and the NVFP4-expert GEMM - are llama
|
||||
**wins** after this series (the recurrence runs at 102.6% of vLLM's bandwidth;
|
||||
the GEMM ties vLLM at the LPDDR5x BW floor). The residual gap is **bf16-projection
|
||||
bandwidth + the host scheduling loop**, both at the LPDDR5x floor - not a kernel
|
||||
llama is losing. The MoE GEMM kernel is *not* where the gap lives.
|
||||
|
||||
**Rejected / flat levers** (recorded so they are not re-tried):
|
||||
|
||||
- **Lever 2 - graph/stream coverage: FLAT.** Bit-exact graph coverage was
|
||||
exhausted by 0025; more graph/stream overlap is a no-op or small regression on
|
||||
this model.
|
||||
- **D1 premise "static decode is host-sync-bound on the MoE-routing readback":
|
||||
REFUTED.** The hypothesis was that the dominant decode cost is the device->host
|
||||
readback of MoE routing before launching the per-expert GEMMs (mul_mat_id's
|
||||
per-expert host-loop fallback). Profiling (GB10, q36-35b-a3b-nvfp4, batched-bench
|
||||
npl128) shows the opposite: on NVFP4 the grouped stream-k MMQ id-path is what
|
||||
runs (routing stays device-side), so the host-loop fallback is **never hit** -
|
||||
`cudaStreamSynchronize` count is *identical* with CUDA graphs on vs off (1457
|
||||
either way; only the kernel-launch count changes, ~100k vs ~229k). Steady-decode
|
||||
GPU-busy is **~99%** (1% idle), i.e. static decode is GPU-bound, not idle waiting
|
||||
on a sync. The one actionable residual the profile surfaced - per-step host
|
||||
kernel **re-issue** when the step is not graph-captured - shipped as 0043
|
||||
(default-on full-step decode graph), worth +2.6% (npl128) to +5-13% (npl32). The
|
||||
larger continuous-serving host cost is the graph **rebuild** (0040/0041), and the
|
||||
irreducible floor is the per-step logits-D2H-before-sampling serial point - none
|
||||
of which is the MoE-routing readback.
|
||||
- **Lever 3 - act-quant fusion: FLAT.** The W4A4 act-quant tax is removable only
|
||||
by W4A16 (a precision change, rejected) or a structural kernel rewrite; no
|
||||
further bit-exact lever clears it. 0023 already banks the de-dup.
|
||||
- **Lever 4 - NVFP4 the bf16 GDN/attn projections: REJECTED (KL-gate fail).**
|
||||
Quantizing the projections to NVFP4 costs ~+6% PPL; vLLM deliberately keeps the
|
||||
same bf16 projections. No-ship.
|
||||
- **W4A16-Marlin MoE GEMM: REJECTED.** It would be a precision upgrade nobody
|
||||
needs bought with a ~5% slower kernel; both kernels are already at the BW floor.
|
||||
(The "the win was NVFP4-dense-quant, not the Marlin kernel" dense verdict
|
||||
carries over to MoE.)
|
||||
- **Chunked parallel-scan GDN prefill (patch 0031): the scalar-serial form was
|
||||
FLAT-to-SLOWER at C=16 - the tensor-core M5 form (patch 0047) is the win,
|
||||
now DEFAULT-ON under paged KV.** 0031 implements the upstream "faster pre-fill"
|
||||
TODO - the FLA-style chunked gated-delta-rule (intra-chunk delta rule solved in
|
||||
parallel via the UT-transform + forward substitution, inter-chunk recurrence
|
||||
over n_tokens/C steps), math validated equivalent (numpy f32 NMSE ~1e-13;
|
||||
`test-backend-ops` within the 1e-7 NMSE gate, a NEW per-path result). **But
|
||||
GB10's 99KB dynamic-smem opt-in forces C=16** (the 128x128 f32 state alone is
|
||||
64KB of the all-shared layout); the scalar-serial scan (`GDN_TC=0`) was then
|
||||
pinned to 1 block/SM with serial per-thread dk-reductions and measured **~761
|
||||
t/s chunked vs ~971 t/s sequential (~22% slower)**, grid-starved at low n_seqs.
|
||||
The lesson held: **at this head dim the win needs tensor cores, not just
|
||||
chunking.** Patch 0047 builds those tensor-core forms (KK/QK Gram = M2, KS/QS
|
||||
state-boundary 3xtf32 = M3, P*U output = M4, full form-T solve + state-update
|
||||
mma = M5, all `GDN_TC`-selected in one build) and ships **M5** as the default
|
||||
when `LLAMA_KV_PAGED` is set. It is an f32/tf32-only re-port: the bf16/hybrid
|
||||
dev-tree machinery (from the dropped 0026 ssm_bf16_tau) and the bf16 CONFIG-C
|
||||
(M8) plus register-resident M6/M7 variants are NOT part of this series. M5 is the
|
||||
variant that beats the (already 84.7%-of-peak) sequential scan while staying on
|
||||
the bit-exact gate: MoE prefill S_PP **+3.5% @npp512 (3x interleaved A/B), +17.7%
|
||||
@npp2048**; decode S_TG unchanged (the tuned `GDN_CHUNK_MIN=64` engage threshold
|
||||
is > 1, so the 1-tok decode steps never enter the chunked path - at
|
||||
`GDN_CHUNK_MIN=1` the chunked path swallows decode and collapses S_TG ~25%, the
|
||||
reason the threshold is the lever). Bit-exactness is per-path benign:
|
||||
`test-backend-ops` GATED_DELTA_NET is **94/94** vs CPU with M5 forced (incl.
|
||||
multi-chunk n_tokens up to 256); the greedy md5 default-on == M5-forced ==
|
||||
canonical on the short gate prompt (paged-MoE `8cb0ce23`, dense `5951a5b4`); on
|
||||
a long MoE prompt (where the default fires M5 at >=64 tokens) M5 and the
|
||||
sequential path agree word-for-word until **one** benign greedy token-flip
|
||||
("the User:" vs "the User's Request:"), the dense model not flipping at all -
|
||||
the textbook reduction-order flip greedy amplifies, NMSE-validated. The chunk
|
||||
geometry stays env-selectable (`GDN_TC`/`GDN_CHUNK_C`/`GDN_DV_TILE`) for further
|
||||
tuning; M5 is the shipped default because it wins without losing the canonical gate.
|
||||
- **GDN occupancy retune (patch 0022) was a decode win but an UNCONDITIONAL
|
||||
dense-prefill regression - now gated by scan length (patch 0046).** Patch
|
||||
0022's `(NUM_WARPS=16, COLS_PER_WARP=8)` column-fold of the GDN
|
||||
sequential-recurrence dispatch (`case 128`) raises per-warp memory-level
|
||||
parallelism on the short, wide DECODE scans (small `n_tokens`, large
|
||||
`n_seqs`) - the measured +11.1% dense decode win. Applied unconditionally it
|
||||
also hit the dense PREFILL path, where the scan is long and narrow: the launch
|
||||
`grid.z` collapses from `S_v/4 = 32` to `S_v/(16*8) = 1`, the SMs starve, and
|
||||
profiling attributed the whole ~-6% dense-prefill regression vs stock to
|
||||
`gated_delta_net` (+54% GPU time at the (16,8) geometry). Patch 0046 gates the
|
||||
geometry by per-call scan length: long scans (prefill,
|
||||
`n_tokens >= GDN_PREFILL_NTOK`, default 256) take stock's high-grid.z `(4,1)`
|
||||
geometry; short scans (decode) keep the `(16,8)` retune. That recovers dense
|
||||
prefill +7.2% back to stock parity while keeping the decode win, and it is
|
||||
bit-exact: patch 0022 already proved every selectable `{NUM_WARPS,
|
||||
COLS_PER_WARP}` variant is byte-identical (the sweep cannot change the md5), so
|
||||
switching geometry by scan length cannot move the greedy output. The explicit
|
||||
`GDN_NW`/`GDN_CPW` one-build %peak sweep still overrides (the gate yields when
|
||||
either is set), so the A/B harness is unchanged.
|
||||
|
||||
**Opt-in bf16-SSM fast mode - DROPPED (was patch 0026, `ssm_bf16_tau`).** The
|
||||
design premise - that bf16 KL error concentrates in long-memory heads and can be
|
||||
removed by keeping them f32 - was already shaky: the error scales with the bf16
|
||||
head *count* and saturates (~0.06 MeanKLD / ~91% same-top-p) far below any useful
|
||||
byte saving. The lever was then **removed entirely** once the decode fusions
|
||||
(0028 recurrent-state gather-fusion + 0029 block-table cache) landed: a clean
|
||||
re-measurement that forced **all** gated-DeltaNet heads to bf16 (`tau=100000`,
|
||||
the most aggressive setting) gave **flat** decode throughput - **780.6 vs 780.0
|
||||
t/s**. The mode engages but buys **zero** speed; the earlier "+12%" was subsumed
|
||||
by the fusions. So bf16-tau was a precision trade (not bit-exact) plus extra bug
|
||||
surface and CUDA template-instantiation compile cost with **no** offsetting
|
||||
benefit, and patch 0026 was dropped from the series. Lesson recorded so it is not
|
||||
re-tried: do not reintroduce a per-head SSM-precision lever - the bandwidth it
|
||||
targeted is already recovered by the gather-fusion + block-table cache.
|
||||
|
||||
---
|
||||
|
||||
## 6. Architecture and quant generality
|
||||
|
||||
(From the arch-generality and quant-generality audits.)
|
||||
|
||||
- **15 of 16 optimizations are quant-AGNOSTIC.** Only **0023** (NVFP4
|
||||
activation-quantize de-dup) is NVFP4-specific. The SSM/paged/MMQ optimizations
|
||||
help **any quant** of these models (the GDN recurrence, conv, gather and
|
||||
o_proj-MMQ levers operate on the f32 recurrent state and the routing layout,
|
||||
not on the weight dtype).
|
||||
- **Arch-safe to build everywhere.** NVFP4 use is Blackwell-gated and falls back
|
||||
to dequant on other hardware; the GB10-tuned occupancy params (0022) are
|
||||
perf-only and env-selectable (`GDN_NW` / `GDN_CPW`), so they never change
|
||||
correctness on other GPUs. Patch 0030 makes the fused-op emission CUDA-family +
|
||||
CPU only, so a non-CUDA paged build routes to the safe upstream non-fused path.
|
||||
|
||||
- **What generalizes beyond this backend (upstream candidates).** The *speedups*
|
||||
are CUDA/Blackwell-specific (which is why Metal/Vulkan don't benefit - section
|
||||
4c), but several *findings and ops* are portable and worth upstreaming:
|
||||
- The headline is hardware-independent: on hybrid gated-DeltaNet models, decode
|
||||
is bottlenecked by the recurrent-state **plumbing** (memcpy + gathers, ~67% of
|
||||
the step), not the weight GEMM. The fusions for it (in-place state 0018, gather
|
||||
0019/0028, conv 0021) are bit-exact and already have CPU reference kernels, so
|
||||
they would speed up Qwen3.6 / Qwen3-Next / any hybrid-SSM decode on **every**
|
||||
backend once the ggml ops gain the respective (Metal/Vulkan) kernels - the
|
||||
highest-value upstream contribution.
|
||||
- The o_proj GEMV->MMQ reshape (0020) is a model-graph fix (batch the projection
|
||||
to hit the GEMM path) - arch-agnostic in principle, trivial to upstream.
|
||||
- The paged KV + cross-request prefix sharing + decode-first scheduler align with
|
||||
llama.cpp's own in-progress KV / chunked-prefill work and could inform it.
|
||||
- The per-path bit-exact md5 gate + the weekly upstream-drift canary is a reusable
|
||||
maintenance pattern for any vendored-patch backend.
|
||||
|
||||
---
|
||||
|
||||
## 7. Pin + maintenance policy
|
||||
|
||||
- **Canonical source = the fork branch `mudler/llama.cpp:localai-paged`.** The
|
||||
vendored `patches/paged/*.patch` files are now generated (one `git format-patch`
|
||||
per commit) from that branch, which is the pin commit plus the paged patch
|
||||
commits in order, so there is no more hand-export drift between the dev tree and
|
||||
the shipped series.
|
||||
- **Pinned to llama.cpp `0ed235ea2c17a19fc8238668653946721ed136fd`** (kept == the stock `llama-cpp` pin). The pin
|
||||
is advanced **only** by the manual pin-sync process (this section):
|
||||
rebase the source-only patch series onto the new tip, rebuild on GPU, pass the
|
||||
bit-exact gate on every path (dense + MoE, paged + non-paged) plus
|
||||
`test-backend-ops`, **and confirm the full grpc-server build links on CI**.
|
||||
- **The pin must track the stock pin.** `grpc-server.cpp` is shared with the stock
|
||||
backend and tracks the stock pin, so a paged pin that diverges past an upstream
|
||||
server-API refactor breaks the grpc-server LINK even when the patches are
|
||||
bit-exact. A bump to `c299a92c` (23 commits ahead of stock) was greedy-md5
|
||||
bit-exact but failed to link (undefined `stream_*` server helpers introduced by
|
||||
the refactor), and was reverted to the then-current stock pin. The bit-exact gate alone does not
|
||||
catch this; only the full CI grpc-server build does.
|
||||
- **Decoupled from the nightly auto-bumper.** There is deliberately **no**
|
||||
`bump_deps.yaml` entry for this backend - a naive `LLAMA_VERSION` bump could
|
||||
silently shift the tree out from under the patches.
|
||||
- **Weekly canary.** [`.github/workflows/llama-cpp-paged-canary.yml`](../../../.github/workflows/llama-cpp-paged-canary.yml)
|
||||
(via [`.github/scripts/paged-canary-apply.sh`](../../../.github/scripts/paged-canary-apply.sh))
|
||||
tries the patch series against the latest upstream tip with the build's own
|
||||
strict `git apply`. **Red = upstream drifted past the series -> run a
|
||||
PIN_SYNC** (do not bump the pin blindly), following the policy in this section.
|
||||
|
||||
---
|
||||
|
||||
## 8. Models
|
||||
|
||||
> **Build coverage: CUDA-only.** This backend ships only the CUDA/cublas build
|
||||
> targets (cuda-12, cuda-13, and the nvidia-l4t arm64 cuda-12/cuda-13 Jetson
|
||||
> rows). There are no cpu / vulkan / sycl / hipblas / metal-darwin builds: the
|
||||
> patchset's wins are CUDA/Blackwell-specific (section 4c), so off-CUDA the
|
||||
> backend is neutral-to-negative and non-CUDA users should run the stock
|
||||
> `llama-cpp` backend instead. The `backend/index.yaml` meta-backend resolves
|
||||
> `default`/`nvidia` to a CUDA variant accordingly.
|
||||
|
||||
The benchmarked NVFP4 GGUFs are published and wired into the LocalAI gallery:
|
||||
|
||||
| Gallery entry | Weights (HuggingFace) | Notes |
|
||||
|---|---|---|
|
||||
| `qwen3.6-27b-nvfp4-paged` | [`mudler/Qwen3.6-27B-NVFP4-GGUF`](https://huggingface.co/mudler/Qwen3.6-27B-NVFP4-GGUF) | Dense, native Blackwell NVFP4 (FP4-MMA). |
|
||||
| `qwen3.6-35b-a3b-nvfp4-paged` | [`mudler/Qwen3.6-35B-A3B-NVFP4-GGUF`](https://huggingface.co/mudler/Qwen3.6-35B-A3B-NVFP4-GGUF) | MoE (256 experts, top-8), `file_type MOSTLY_NVFP4`. |
|
||||
|
||||
Both gallery entries set `backend: llama-cpp-localai-paged` and the paged serving config
|
||||
(`paged_kv:true`, `max_batch_tokens`, `kv_unified:false`, `parallel`,
|
||||
`flash_attention:on`, `context_size`). They are bit-exact. The full
|
||||
backend-split + gallery plan is in
|
||||
[`LOCALAI_LLAMACPP_BACKEND_PLAN.md`](docs/LOCALAI_LLAMACPP_BACKEND_PLAN.md).
|
||||
|
||||
---
|
||||
|
||||
## 9. vLLM parity - final state (CLOSED)
|
||||
|
||||
> 2026-07-01 follow-up: the investigation was reopened for MTP safety,
|
||||
> MTP-serving, graph-shape tracing, and a current-stack serving snapshot. Phases
|
||||
> 14-20 are recorded in
|
||||
> [`docs/GB10_PARITY_PHASE0_RESULTS.md`](docs/GB10_PARITY_PHASE0_RESULTS.md) and
|
||||
> [`docs/PARITY_HANDOFF.md`](docs/PARITY_HANDOFF.md). They did not change the
|
||||
> GB10 conclusion: MTP/scheduler shortcuts are rejected, and the latest clean
|
||||
> stack remains below vLLM serving parity.
|
||||
|
||||
The multi-week GB10 (DGX Spark, sm_121) vLLM-parity investigation is **closed**.
|
||||
The standing, never-re-litigate record - full benchmark, every lever and verdict,
|
||||
the structural floors, the parity verdict - is
|
||||
[`docs/VLLM_PARITY_FINAL.md`](docs/VLLM_PARITY_FINAL.md). Summary:
|
||||
|
||||
- **Where we are (GB10, Qwen3.6 NVFP4, vs vLLM 0.23.0).** Decode: dense is
|
||||
**ahead of vLLM at low concurrency (116.7% at N=8)** and both models are
|
||||
bandwidth-floored at **~56-68% of vLLM at high concurrency**. Prefill is
|
||||
**~36% (MoE) / ~43% (dense)** of vLLM. Memory: **1.5-3x lower** than vLLM
|
||||
(NVFP4-resident; vLLM's peak is a fixed ~109-112 GB 0.85-util reservation,
|
||||
paged grows with KV from ~50 GB). Output is bit-exact per-path
|
||||
(`5951a5b4` dense, `8cb0ce23` paged-MoE).
|
||||
- **Why the residual is a hardware ceiling, not missing work.** Decode kernels
|
||||
are already **5.4x more GPU-efficient per token** than vLLM's; the gap is the
|
||||
**LPDDR5x ~273 GB/s** floor. The prefill GEMM is **FP4-MMQ-optimal** (every
|
||||
alternative - 0033 dequant->cuBLAS, 0034 native FP4-MMA, 0035/Marlin W4A16,
|
||||
offline-repack and vLLM-verbatim Marlin - was rejected; bf16 TC peak is ~half
|
||||
FP4 peak, and vLLM itself runs a bf16-Marlin fallback on sm_121). The GDN
|
||||
chunked scan is at the tractable tensor-core win (**M5 tf32**, patch 0047);
|
||||
its residual is the **O(C^2) intra-chunk solve + serial recurrence** (occupancy
|
||||
and dtype proven not the bound: BV -1%, bf16-C64 -18.75%). The serving host
|
||||
loop is **closed** (~0-1% of the wall; padded-decode built + rejected).
|
||||
- **Shipped, bit-exact wins.** FP4-MMQ GEMM, M5 tensor-core GDN prefill (0047),
|
||||
fused residual+RMSNorm (0042), fused GatedRMSNorm+SiLU (0044), GDN-prefill
|
||||
geometry gate (0046), the SSM decode-fusion stack (0018-0022/0028, up to
|
||||
2.46x/2.26x over stock), decode-graph reuse (0040/0043), the memory advantage,
|
||||
and low-N decode lead.
|
||||
- **The path to parity is different hardware.** Datacenter Blackwell (HBM,
|
||||
native tcgen05/CUTLASS FP4) lifts the bandwidth floor and **restores exactly
|
||||
the vLLM advantages that lose on GB10** (FLA blocked-solve GDN, Marlin/CUTLASS
|
||||
grouped FP4, HBM-tuned full-cudagraph decode). Re-run the methodology on new
|
||||
silicon; do not reopen the GB10 levers.
|
||||
|
||||
Latest current-stack MoE serving snapshot (`PTOK=128`, `GEN=64`, current clean
|
||||
DGX mirror `f2521ab12`, artifact
|
||||
`/home/mudler/bench/phase26_audited_snapshot/20260701_053650`). This run
|
||||
includes `hardware.txt` and `gate_summary.tsv`; all pre/post gate rows are
|
||||
`ok`:
|
||||
|
||||
| n | paged decode_agg | vLLM decode_agg | paged/vLLM decode | paged agg | vLLM agg | paged/vLLM agg |
|
||||
|---|------------------|-----------------|-------------------|-----------|----------|----------------|
|
||||
| 8 | 230.8 | 283.2 | 81.5% | 170.6 | 241.6 | 70.6% |
|
||||
| 32 | 420.0 | 609.0 | 69.0% | 254.6 | 466.7 | 54.6% |
|
||||
| 128 | 673.4 | 1025.0 | 65.7% | 324.0 | 656.5 | 49.4% |
|
||||
|
||||
Use `paged-current-serving-snapshot.sh` for future current-stack GB10 serving
|
||||
snapshots. It targets the clean `~/llama-phase6-source` mirror, checks
|
||||
docker/`local-ai-worker`/GPU-idle state, uses the owner-file lock, runs pre/post
|
||||
inference gates, writes `hardware.txt`, emits `gate_summary.tsv`, and emits
|
||||
paged/vLLM ratios.
|
||||
`hardware.txt` records the GPU identity and hardware class so GB10/workstation
|
||||
Blackwell evidence is not confused with a future datacenter-Blackwell rerun.
|
||||
`gate_summary.tsv` records pre/post MoE md5, dense md5, and backend-op checks
|
||||
so an artifact proves inferencing gates without reading full logs.
|
||||
Do not use the stale DGX
|
||||
`~/bench/combined_definitive.sh` without first porting it to the current mirror
|
||||
and lock discipline.
|
||||
|
||||
Phase 28 challenged the remaining low-conflict NVFP4 grouped-MMQ occupancy
|
||||
knobs on the same DGX mirror
|
||||
(`/home/mudler/bench/phase28_mmq_occupancy/20260701_040450`). The only buildable
|
||||
variant, `GGML_CUDA_FP4_MINBLOCKS=2`, was inference-safe before and after
|
||||
serving (MoE `8cb0ce23`, dense `5951a5b4`, `MUL_MAT_ID 806/806`) but regressed
|
||||
n128 decode serving (`705.1 -> 689.9` decode_agg_tps, `0.9784x`). The row-tile
|
||||
knob `GGML_CUDA_FP4_MMQ_Y=64` failed the NVFP4 writeback compile-time
|
||||
invariant. Do not promote these knobs; grouped-MMQ parity work now requires a
|
||||
structural kernel change, not launch-bounds or row-tile tweaks.
|
||||
|
||||
Phase 29 added the default-off grouped-MMQ shape trace as patch `0056`
|
||||
(`/home/mudler/bench/phase29_mmq_shape_trace/20260701_042428`). The helper was
|
||||
added test-first (`test-cuda-mmq-shape-trace`), compiled under CUDA on DGX, and
|
||||
kept inference stable with the trace disabled and enabled:
|
||||
MoE `8cb0ce23`, dense `5951a5b4`, `MUL_MAT_ID 806/806`. Example trace line:
|
||||
`[LLAMA_MOE_MMQ_SHAPE] type=40 moe=1 ncols_dst=104 nchannels_x=256 ncols_max=13 n_active_est=104 density=1 mmq_x_max=128 mmq_x_lim=64 mmq_x_best=16 mmq_y=128 stream_k=1`.
|
||||
|
||||
Phase 31 extended that trace as patch `0057`
|
||||
(`/home/mudler/bench/phase31_mmq_launch_trace/20260701_064424`) with
|
||||
`[LLAMA_MOE_MMQ_LAUNCH]` lines from `launch_mul_mat_q`. Default-off,
|
||||
trace-enabled, and post-serving gates stayed stable: MoE `8cb0ce23`, dense
|
||||
`5951a5b4`, `MUL_MAT_ID 806/806`. The n128 serving trace showed decode-like
|
||||
`4800/4800` and prefill-like `4920/4920` launch lines with `fixup=0` and
|
||||
`stream_k_blocks == ntiles_dst`, rejecting a no-fixup/no-stream-k shortcut for
|
||||
this workload.
|
||||
|
||||
Phase 32 added the small-M classifier trace as patch `0058`
|
||||
(`/home/mudler/bench/phase32_small_m_classifier/20260701_070127`). Default-off,
|
||||
trace-enabled, and post-serving gates stayed stable: MoE `8cb0ce23`, dense
|
||||
`5951a5b4`, `MUL_MAT_ID 806/806`. The n128 serving trace found 4096 small-M
|
||||
candidate calls: `mmq_x_best=64` 1800, `48` 1096, `40` 360, `32` 360, `16`
|
||||
360, `24` 120. This justifies Phase 33 as a default-off tile-policy A/B
|
||||
(`mmq_x=16`, possibly `8`) rather than a broad kernel rewrite.
|
||||
|
||||
Phase 33 added default-off `LLAMA_MOE_SMALL_M_TILE=<n>` as patch `0059`
|
||||
(`/home/mudler/bench/phase33_small_m_tile_policy/20260701_071136`). The knob is
|
||||
md5/op safe, but both tested values were slower in same-session n128 serving:
|
||||
baseline `672.1` decode_agg_tps, tile16 `640.3` (`0.953x`), tile8 `583.2`
|
||||
(`0.868x`). Do not promote simple smaller `mmq_x` caps for this workload.
|
||||
|
||||
Phase 34 added default-off `LLAMA_MOE_MMID_ROUTE_TRACE=<n>` as patch `0060`
|
||||
(`/home/mudler/bench/phase34_mmid_route_trace/20260701_072737`). Default-off,
|
||||
trace-enabled, and post-serving gates stayed stable: MoE `8cb0ce23`, dense
|
||||
`5951a5b4`, `MUL_MAT_ID 806/806`. Live n128 serving with trace cap 4096 produced
|
||||
`mmq=2776`, `mmvq=1320`, and `host_sync=0/4096`; the top shapes were
|
||||
`mmq ne2=12` (1096), `mmq ne2=18` (480), and `mmvq ne2=8` (360). This refutes
|
||||
host-sync fallback as the current n128 `MUL_MAT_ID` problem; follow-up work should
|
||||
target grouped-MMQ small-M kernel partitioning or another measured bucket.
|
||||
|
||||
Phase 35 added default-off `LLAMA_MUL_MAT_ROUTE_TRACE=<n>` as patch `0061`
|
||||
(`/home/mudler/bench/phase35_mul_mat_route_trace/20260701_074359`). Default-off,
|
||||
trace-enabled, and post-serving gates stayed stable: MoE `8cb0ce23`, dense
|
||||
`5951a5b4`, `MUL_MAT 1146/1146`, `MUL_MAT_ID 806/806`. Live n128 serving with
|
||||
trace cap 8192 produced route counts: `mat_f=2888`, `op_cublas=2292`,
|
||||
`mmq=1328`, `vec_q=1214`, `vec_f=470`. BF16 (`type=30`) dominated the trace
|
||||
with `mat_f=2485` and `op_cublas=1330`; top BF16 shapes were `mat_f ne1=12`
|
||||
(775), `op_cublas ne1=18` (760), and `mat_f ne1=8` (570). Next projection work
|
||||
should trace or optimize the BF16 `op_cublas`/`mat_f` split, not batched cuBLAS.
|
||||
@@ -1,374 +0,0 @@
|
||||
# Accelerator-porting scope: bringing the paged backend's portable benefits to Metal / SYCL / Vulkan (+ a ROCm note)
|
||||
|
||||
Source-only analysis (no GPU, no build) of which `llama-cpp-localai-paged` benefits
|
||||
are portable off the CUDA family, and what each port costs per accelerator. This is
|
||||
the umbrella doc; it BUILDS ON, and does not repeat,
|
||||
[`UPSTREAM_LAYER2_SCOPE.md`](UPSTREAM_LAYER2_SCOPE.md) (the GDN/SSM fusion kernel
|
||||
scope) - that doc remains the authoritative reference for benefit #1 below.
|
||||
|
||||
The backend ships **CUDA-only** today (README sections 4c, 8): off-CUDA the fusions
|
||||
gate off (patch 0030) and NVFP4 falls back to dequant, so it is
|
||||
neutral-to-slightly-negative there and non-CUDA users run the stock `llama-cpp`.
|
||||
"Porting the benefits" is the upstream-contribution track that would make these
|
||||
wins real on the other accelerators. Methodology for the work itself is in
|
||||
[`.agents/vllm-parity-methodology.md`](../../../../.agents/vllm-parity-methodology.md).
|
||||
|
||||
We have **no Metal / SYCL / Vulkan / ROCm hardware here**, so every port is gated
|
||||
by `test-backend-ops` (backendX-vs-CPU) **on the target hardware** - the same gate
|
||||
discipline the existing layer-2 doc sets out.
|
||||
|
||||
--------------------------------------------------------------------------------
|
||||
## 0. The four benefits and their portability class
|
||||
|
||||
| # | Benefit (patches) | Portable off CUDA? | Where scoped |
|
||||
|---|---|---|---|
|
||||
| 1 | **GDN/SSM decode fusions** (0018-0022, 0028) - in-place state write-back, fused recurrent-state gather, conv-state in-place fusion, o_proj MMQ reshape, occupancy retune | YES - per-backend KERNEL work | [`UPSTREAM_LAYER2_SCOPE.md`](UPSTREAM_LAYER2_SCOPE.md) (consolidated in section 1 here) |
|
||||
| 2 | **Paged KV in-kernel block-table flash-attn read** (0009-0011) | YES - per-backend KERNEL work | **Section 2 here (the new analysis)** |
|
||||
| 3 | **Decode-first prefill scheduler** (0013/0016) | YES - FREE, host-side, zero kernel work | Section 3 here |
|
||||
| 4 | **NVFP4 FP4-MMA + its decode levers** (0017/0023/0025) | NO (Blackwell FP4-MMA) - out of scope; two analogues flagged | Section 4 here |
|
||||
|
||||
The two kernel-bearing tracks (#1 and #2) share an identical port SHAPE - they touch
|
||||
the same decode kernel(s), the same `supports_op`, the same dispatch guard, and
|
||||
sequence the same way (ops-first PR, then one PR per backend). They should be
|
||||
**bundled into one per-backend PR**, not pursued as two separate efforts; section 5
|
||||
sequences them together. Tracks #3 (free) and #4 (out of scope) are independent.
|
||||
|
||||
--------------------------------------------------------------------------------
|
||||
## 1. Benefit #1 - GDN/SSM decode fusions (consolidated; full scope is the layer-2 doc)
|
||||
|
||||
Do not re-derive this here. [`UPSTREAM_LAYER2_SCOPE.md`](UPSTREAM_LAYER2_SCOPE.md)
|
||||
already establishes, and this doc adopts wholesale:
|
||||
|
||||
- The base `GGML_OP_GATED_DELTA_NET` + `GGML_OP_SSM_CONV` + `GGML_OP_SSM_SCAN`
|
||||
kernels **already exist on Metal, Vulkan AND SYCL**, so the Qwen3.6 hybrids RUN
|
||||
on all three today via the upstream non-fused path. Layer-2 is the decode
|
||||
SPEEDUP, not "make it run." (NB: the README section 4c no longer carries the
|
||||
stale "no Vulkan kernel" line that the layer-2 doc section 0 was written to
|
||||
correct - that correction has since been folded into the README, so treat
|
||||
layer-2 section 0 as historical context, not a live correction.)
|
||||
- The four fusion ops (A in-place state 0018, B fused state gather 0019, C
|
||||
conv-update in-place 0021, D conv-tap gather 0028) reuse the existing op enums
|
||||
with extra `src[]` discriminators; only OP C is a genuinely new kernel, the rest
|
||||
redirect the read source / write target of the EXISTING kernel. The builders,
|
||||
CPU reference kernels, model graph and `test-backend-ops` cases are SHARED and
|
||||
already done.
|
||||
- Per-backend net-new work, effort and gotchas: **SYCL easiest** (near-verbatim
|
||||
CUDA mirror, ~250-350 LOC, no shader-gen), **Metal medium** (~350-500 LOC,
|
||||
fixed 32 simdgroup = simplest bit-exactness), **Vulkan hardest** (~450-650 LOC +
|
||||
shaders-gen + descriptor growth + per-vendor subgroup validation).
|
||||
- Bit-exactness is per-backend BY CONSTRUCTION (the fusions redirect addresses, not
|
||||
the f32 reduce order); gated by `test-backend-ops` (backendX-vs-CPU).
|
||||
- Upstream path: ops-first PR (incl. the capability-driven replacement for patch
|
||||
0030's backend-name allow-list), then one PR per backend.
|
||||
|
||||
The value/effort ranking from that doc (**Metal 1st, SYCL 2nd, Vulkan 3rd**) is
|
||||
adopted unchanged here and, as section 5 shows, coincides with benefit #2's ranking
|
||||
- which is why the two bundle cleanly per backend.
|
||||
|
||||
--------------------------------------------------------------------------------
|
||||
## 2. Benefit #2 - paged KV in-kernel block-table flash-attn read (NEW scope)
|
||||
|
||||
### 2.0 What it is, and why it is the lever that makes paged KV non-negative off-CUDA
|
||||
|
||||
On CUDA, patches 0009-0011 replaced the per-step host-side K/V gather (patch 0003)
|
||||
with an **in-kernel paged read**. `ggml_flash_attn_ext` gained an optional
|
||||
`src[5]` = an I32 block table `[n_view, n_stream]` in token-POSITION order; the
|
||||
fattn vec/tile kernel maps logical KV index `j` to physical cell
|
||||
`block_table[seq*ne11 + j]` and reads `K0 + cell*nb11` / `V0 + cell*nb21` in place,
|
||||
so the `get_rows` of K and V (the bulk of the gather) is gone. A null block table is
|
||||
the stock contiguous read, byte-identical. Position ordering keeps the online-softmax
|
||||
reduction order identical to stock, so it is bit-exact (CPU/batch1) by construction.
|
||||
|
||||
The crucial point for portability: **the entire host side is already
|
||||
backend-agnostic.** The block-table fill (`llama_kv_cache::get_block_table`), the
|
||||
K/V views, the mask compaction, the `input_block_table` graph input, and the
|
||||
`ggml.c` / `ggml.h` builder (`ggml_flash_attn_ext_set_block_table`) all live in
|
||||
`src/` and `ggml/...` shared code. The ONLY per-backend work is, in each backend's
|
||||
flash-attn kernel: (a) thread one extra source through to the kernel, and (b) do the
|
||||
indexed read at the K/V load sites. The CPU reference already does it (patch 0009,
|
||||
`ops.cpp`).
|
||||
|
||||
Off-CUDA today the paged path falls back to the **host-side gather** (patch 0003),
|
||||
which the README section 4c measured as neutral-to-slightly-negative on the M4
|
||||
(~0-3% slower decode / ~2-8% slower prefill vs stock's contiguous read - pure
|
||||
overhead, because the in-kernel read that *recovers* the gather cost is CUDA-only).
|
||||
**Porting the block-table read is exactly what flips paged KV from
|
||||
"neutral-to-negative" to "neutral-to-positive" off CUDA** - it removes the gather
|
||||
overhead so paged KV's memory-management and prefix-sharing wins come for free
|
||||
instead of at a decode tax. (The big decode multipliers on the hybrids are still the
|
||||
benefit-#1 GDN fusions; this benefit is what makes the paged *allocator* pay its own
|
||||
way off CUDA.)
|
||||
|
||||
### 2.1 The cross-cutting finding (applies to all three backends)
|
||||
|
||||
The indexed per-cell read only fits the **vec / scalar decode kernel**. Every
|
||||
backend's FAST attention path - CUDA mma, Metal `simdgroup_load` MM, Vulkan
|
||||
coopmat2, SYCL tile - loads K/V as **contiguous tiles** (8-cell `simdgroup_load`,
|
||||
`coopMatLoadTensorNV` over a linear stride, shared-memory tile loads) that cannot
|
||||
express an arbitrary per-cell gather without a staging pre-pass. This is exactly why
|
||||
the CUDA port (0009-0010) wired ONLY the vec kernel and added a dispatch guard
|
||||
(`if (dst->src[5]) force vec`).
|
||||
|
||||
So each port mirrors that: **route any FA op carrying a block table onto the vec /
|
||||
scalar kernel; leave the fast MM path contiguous-only**, and keep the null-table
|
||||
contiguous read on the fast path untouched. The decode shape (1 query token/stream)
|
||||
naturally lands on or near the vec/scalar kernel on all three, so this is a small
|
||||
routing change, not a rewrite of the fast path.
|
||||
|
||||
### 2.2 SYCL - EASIEST (near line-for-line CUDA mirror)
|
||||
|
||||
- **Exists today:** `ggml-sycl/fattn-vec.hpp` is a DPCT-style near-verbatim mirror
|
||||
of CUDA `fattn-vec.cuh`; the kernel signature ends in the same `nb11..nb33`
|
||||
cluster the CUDA patch appends `const int* block_table` to (fattn-vec.hpp:65-76).
|
||||
Args are passed by SYCL lambda value-capture - **no descriptor/binding/push-
|
||||
constant bookkeeping at all** (strictly easier than CUDA). `supports_op`
|
||||
(`fattn.cpp` -> `ggml_sycl_get_best_fattn_kernel`) needs no change to ACCEPT
|
||||
`src[5]`.
|
||||
- **Port shape (value: medium / effort: LOW):** append `const int* block_table`
|
||||
to the kernel + `fattn_kernel_t` typedef + `lauch_kernel`/`launch_fattn`
|
||||
(sourcing `dst->src[5]->data`); 3 read-site substitutions (K at line 318, V at
|
||||
389 and 410): `K0 + block_table[seq*ne11 + k_VKQ_0 + i_KQ]*nb11`.
|
||||
- **Two SYCL-specific gotchas:**
|
||||
1. **Pointer pre-advance.** The vec kernel advances `K`/`V` by `k_VKQ_0` OUTSIDE
|
||||
the inner read (fattn-vec.hpp:293-300), so `i_KQ`/`k` are tile-local. The port
|
||||
must keep an UN-advanced base `K0`/`V0`, drop the per-iteration `K +=`/`V +=`
|
||||
on the paged path, and reconstruct the absolute cell. Get this wrong and you
|
||||
read the wrong cells with NO compile error.
|
||||
2. **Dispatch guard is bigger than CUDA's.** f16-GQA decode routes to the TILE
|
||||
kernel, not vec (`fattn.cpp:198-208` fall-through). Add
|
||||
`if (dst->src[5]) return BEST_FATTN_KERNEL_VEC;` near the top of
|
||||
`ggml_sycl_get_best_fattn_kernel`. The shared `fattn_kernel_t` typedef means
|
||||
the tile kernel must gain a matching ignored `block_table` param (or split the
|
||||
typedef) - a trivial chore.
|
||||
- **Bit-exact:** sub-group width (16) is fixed and the indexed read does not touch
|
||||
lane assignment, loop bounds, or the XOR-reduction stride - reduction order is
|
||||
invariant, so the paged vec path is byte-identical to SYCL's own contiguous vec
|
||||
path. Gate: `test-backend-ops` FLASH_ATTN_EXT (with a block-table case) on Intel
|
||||
GPU.
|
||||
|
||||
### 2.3 Metal - EASY-MEDIUM (decode already routes to the vec kernel)
|
||||
|
||||
- **Exists today:** decode (1 query token/stream, GQA) dispatches to
|
||||
`kernel_flash_attn_ext_vec` (`ggml-metal-ops.cpp` `..._use_vec`: `ne01 < 20`).
|
||||
Metal IS a true vec-equivalent (not a single unified FA kernel), and the vec
|
||||
kernel's quantized K/V branches ALREADY compute a per-cell base address
|
||||
(`k + ((ic + NE*cc + ty)*nb11)`, ggml-metal.metal:6934 / V at :7045) - so a
|
||||
per-cell indexed read is unambiguously admissible. `supports_op`
|
||||
(`ggml-metal-device.m` FLASH_ATTN_EXT) inspects no src count, so `src[5]` is
|
||||
accepted as-is.
|
||||
- **Port shape (value: HIGH / effort: EASY-MEDIUM):** append a
|
||||
`device const char * block_table` param after `dst` (**buffer index 8** for vec)
|
||||
+ a kargs field + a `has_block_table` function-constant; reuse the existing
|
||||
"bind dummy when null" idiom for a missing table; substitute the cell index with
|
||||
`block_table[seq*ne11 + cell]` at the K reads (lines 6919/6934) and V reads
|
||||
(7032/7045) - a localized rewrite of ~2 loops (the fast path must adopt the
|
||||
per-cell base form the quantized branch already uses).
|
||||
- **Gotcha:** the **non-vec MM kernel is a HARD blocker** -
|
||||
`simdgroup_load(..., NS10, ...)` reads 8 physically-CONTIGUOUS KV cells as one
|
||||
matrix tile (lines 6160 / 6339-6363); an arbitrary gather can't be a single
|
||||
strided matrix load. Mitigate exactly as CUDA did: force any block-table op onto
|
||||
the vec kernel in `..._use_vec` (ggml-metal-ops.cpp:2517); leave the MM path
|
||||
contiguous-only. Also watch a NAME COLLISION: `kernel_flash_attn_ext_blk` is an
|
||||
existing mask-skip optimization, NOT a paged block table.
|
||||
- **Bit-exact:** fixed 32-wide simdgroup + address-only redirect = byte-identical to
|
||||
Metal's own vec contiguous path. Gate: `test-backend-ops` on Apple Silicon.
|
||||
|
||||
### 2.4 Vulkan - MEDIUM (the fast NVIDIA decode path cannot do it)
|
||||
|
||||
- **Exists today:** three FA shaders - `flash_attn.comp` (scalar/vec),
|
||||
`flash_attn_cm1.comp` (coopmat1, stages K/V through shared memory),
|
||||
`flash_attn_cm2.comp` (coopmat2, the fast NVIDIA path). FA uses **7 descriptor
|
||||
bindings (0-6)**; `supports_op` (`ggml-vulkan.cpp` FLASH_ATTN_EXT) checks
|
||||
specific srcs only, no count check; but `src[5]` is **not even threaded today** -
|
||||
`ggml_vk_flash_attn` stops at `src[4]` (ggml-vulkan.cpp:14537), so wiring it
|
||||
through is part of the work.
|
||||
- **Port shape (value: HIGHEST breadth / effort: MEDIUM):** add binding 7 in the
|
||||
shader(s), bump `7`->`8` in the three `ggml_vk_create_pipeline` calls (:3997,
|
||||
:4033, :4070) and the two dispatch subbuffer lists (passing a dummy when null),
|
||||
and wrap the indexed read in one `phys_kv()` helper applied at the ~4 K + 2 V
|
||||
load sites (flash_attn.comp; the logical index is the same `(j*Bc + ...)`
|
||||
expression at every site).
|
||||
- **Two gotchas, one structural:**
|
||||
1. **Push constants are FULL.** `vk_flash_attn_push_constants` is exactly
|
||||
128 bytes with a `static_assert(... <= 128)` (the Vulkan guaranteed minimum) -
|
||||
**no room for a new field.** Signal "block-table enabled" via the existing
|
||||
`Flags` spec constant (flash_attn_base.glsl, `constant_id=10`, already
|
||||
bit-packed) - add a `BLOCK_TABLE_ENABLE` bit. The per-seq stride is already
|
||||
`p.KV`; the seq index is derivable in `init_indices()`.
|
||||
2. **coopmat2 (the fast NVIDIA GQA-decode path) is INCOMPATIBLE.** Its K/V load
|
||||
is a hardware `coopMatLoadTensorNV` over a LINEAR stride
|
||||
(flash_attn_cm2.comp:307-313/377-383); the decode callback only dequantizes,
|
||||
it cannot remap the physical address. The indexed read drops cleanly into
|
||||
**scalar** (which non-GQA decode already uses) and **cm1** (which stages
|
||||
through shmem - remap the staging loop), but **not cm2**. With a block table
|
||||
present, NVIDIA GQA decode falls back to scalar/cm1 (slower than cm2, still
|
||||
correct); the **null-table path keeps using cm2 unchanged**. AMD/Intel (no
|
||||
cm2) are fully covered by scalar/cm1.
|
||||
- **Net positive?** Yes. Non-GQA decode already runs scalar (paged read ~free);
|
||||
AMD/Intel covered; only NVIDIA GQA decode trades cm2 for scalar/cm1 *when a table
|
||||
is supplied*, and paged KV's payoff is allocator/memory + prefix-sharing, not raw
|
||||
FA throughput, so the trade is contained and the fast contiguous path is
|
||||
untouched.
|
||||
- **Bit-exact:** the read is a per-thread scalar load, subgroup-size agnostic
|
||||
(already abstracted via the `SubGroupSize` spec constant); position ordering keeps
|
||||
the reduction order identical, so byte-identical to the backend's own
|
||||
scalar/cm1 contiguous path. **Build burden is low** - these are EXISTING shader
|
||||
variants recompiling (no new `string_to_spv` shape), so no shaders-gen matrix
|
||||
growth. Gate: `test-backend-ops` per vendor (AMD + Intel + NVIDIA).
|
||||
|
||||
### 2.5 Benefit-#2 ranking and the shared dispatch/supports_op pattern
|
||||
|
||||
| backend | value | author effort | structural risk | rank |
|
||||
|---|---|---|---|---|
|
||||
| SYCL | medium (Intel GPU) | **LOW** (line-for-line; no bindings) | low (pointer pre-advance; force-vec guard) | easiest |
|
||||
| Metal | **HIGH** (largest non-CUDA base) | EASY-MEDIUM (decode = vec already) | medium (MM blocker -> force vec) | mid |
|
||||
| Vulkan| **HIGHEST breadth** (AMD+Intel+NVIDIA) | MEDIUM (7->8 bindings; Flags bit) | medium (cm2 can't; full push-const) | hardest |
|
||||
|
||||
Common to all three (mirrors CUDA 0009-0010): (1) `supports_op` needs no change to
|
||||
ACCEPT `src[5]`; (2) a **dispatch guard forces any block-table op onto the
|
||||
vec/scalar kernel**; (3) the fast MM/coopmat2 path stays contiguous-only and the
|
||||
null-table read on it is byte-identical to stock.
|
||||
|
||||
--------------------------------------------------------------------------------
|
||||
## 3. Benefit #3 - decode-first prefill scheduler (FREE portable win, confirmed)
|
||||
|
||||
Patches 0013 (static `LLAMA_PREFILL_BUDGET`) and 0016 (dynamic decode-first
|
||||
`max(n_ubatch, T-D)`) are **pure host-side scheduler policy inside `update_slots()`
|
||||
with zero libllama / zero ggml-backend changes** (README sections 2, 3). They change
|
||||
only the *count* of prefill tokens admitted per step; they touch no kernel, no
|
||||
`supports_op`, no device code. They are therefore **already backend-portable with no
|
||||
per-accelerator work** - they run identically on Metal, SYCL, Vulkan, ROCm, CPU.
|
||||
Byte-identical when off (default-off / short prefill == upstream `-b` chunking).
|
||||
|
||||
This is the cheapest portable benefit: it needs no port at all, only the decision to
|
||||
leave it enabled in the (currently CUDA-only) build, or to upstream the policy. The
|
||||
only reason it is not "live everywhere" today is that the backend ships CUDA-only;
|
||||
the code itself is accelerator-neutral. If the scheduler levers are upstreamed
|
||||
independently of the kernels, they help any llama.cpp build on any accelerator at
|
||||
once - the lowest-effort, broadest-reach contribution of the whole series.
|
||||
|
||||
--------------------------------------------------------------------------------
|
||||
## 4. Benefit #4 - NVFP4 FP4-MMA (NOT portable) + two backend-agnostic analogues
|
||||
|
||||
The NVFP4 decode track is **Blackwell-specific and out of scope** for accelerator
|
||||
porting: Metal, SYCL, Vulkan and ROCm/AMD lack native FP4-MMA (Metal `supports_op`
|
||||
already excludes NVFP4 from `MUL_MAT`/`MUL_MAT_ID`/`GET_ROWS`; on non-Blackwell the
|
||||
FP4 path dequants). Patch 0017 (dense FP4-GEMM occupancy tune) ships only as the
|
||||
parity gate + default-off instrumentation even on CUDA, so there is nothing to port.
|
||||
|
||||
Two of the NVFP4 *decode levers*, however, have backend-agnostic analogues worth a
|
||||
note (do not over-claim - these are observations, not scoped ports):
|
||||
|
||||
- **0023 (NVFP4 activation-quantize de-dup)** - the IDEA generalizes, the patch does
|
||||
not. The MoE broadcast up/gate projections re-quantize the same token activation
|
||||
once per expert; 0023 quantizes the unique activations once and byte-copies them
|
||||
into the expert-gathered layout. Any backend whose MoE path requantizes a shared
|
||||
activation per-expert (e.g. a Q8 activation-quant before an integer-dot MoE GEMM)
|
||||
could dedup the same way. It is NOT NVFP4-specific in PRINCIPLE - but it IS the
|
||||
one quant-specific patch in the series (README section 6), so a port is a
|
||||
per-backend MoE-quant investigation, not a lift-and-shift. Low priority.
|
||||
- **0025 (MoE decode re-graph / `LLAMA_MOE_FORCE_GRAPHS`)** - keeping the graph/
|
||||
capture path on across the grouped-MMQ MoE decode step is a CUDA-graphs concept.
|
||||
Metal/Vulkan/SYCL have their own command-buffer/graph reuse machinery; the
|
||||
generalizable finding is "the grouped MoE decode step has no host sync, so it is
|
||||
safe to keep in a captured/replayed command buffer." Whether each backend's graph
|
||||
layer already covers this is a per-backend question. The methodology note (README
|
||||
dev notes: graph/stream coverage was a FLAT lever beyond 0025 on CUDA) is the
|
||||
more durable takeaway - do not expect a large graph-coverage win on any backend.
|
||||
|
||||
Neither analogue is on the critical path; both are recorded so the next person does
|
||||
not mistake them for free ports.
|
||||
|
||||
--------------------------------------------------------------------------------
|
||||
## 5. Combined sequencing and top recommendations
|
||||
|
||||
Benefits #1 (GDN fusions) and #2 (block-table FA read) share the port shape
|
||||
(vec/scalar decode kernel + `supports_op`/dispatch guard + ops-first-then-per-backend
|
||||
PR) and rank in the SAME order per backend. So sequence them TOGETHER, per backend,
|
||||
behind one shared ops-first PR:
|
||||
|
||||
1. **PR #1 - OPS (largely done, upstreamable as-is):** the `ggml.h`/`ggml.c`
|
||||
builders, the CPU reference kernels, the CUDA kernels, the `test-backend-ops`
|
||||
cases (GDN fusions AND a FLASH_ATTN_EXT block-table case), and the
|
||||
**capability-driven gate** replacing patch 0030's backend-name allow-list (make
|
||||
`supports_op` + the dispatch guard authoritative, so routing falls out of the
|
||||
normal scheduler fallback and no backend name is hard-coded). Independently
|
||||
mergeable.
|
||||
2. **PR #2 - Metal:** GDN fusion kernels (layer-2 doc) + block-table read into
|
||||
`kernel_flash_attn_ext_vec` + the force-vec routing guard. Gate on Apple Silicon.
|
||||
3. **PR #3 - SYCL:** the near-verbatim CUDA mirror of both tracks + the force-vec
|
||||
guard. Gate on Intel GPU.
|
||||
4. **PR #4 - Vulkan:** GDN fusion shaders + the scalar/cm1 block-table read (cm2
|
||||
stays contiguous, falls back when a table is present) + the `Flags` spec-constant
|
||||
bit + the 7->8 binding bump. Gate per vendor.
|
||||
|
||||
Do NOT bundle the backends into one PR (each needs its own hardware for
|
||||
`test-backend-ops`; reviewers are backend-specialized; a regression in one must not
|
||||
block the others).
|
||||
|
||||
### Top recommendations
|
||||
|
||||
1. **Metal first, both benefits together.** Largest non-CUDA LocalAI base; the
|
||||
decode shape already routes to the Metal vec kernel (block-table read is
|
||||
EASY-MEDIUM there) and the base GDN/conv kernels already exist (fusions are
|
||||
MEDIUM); fixed 32-wide simdgroup makes bit-exactness the simplest of the three.
|
||||
Highest value at moderate effort.
|
||||
2. **SYCL second as the cheap mechanical follow-on.** Both tracks are near
|
||||
line-for-line CUDA mirrors with no binding/shader-gen bookkeeping, so it is
|
||||
low-cost insurance even though the Intel-GPU audience is smaller. Budget the
|
||||
effort on the two SYCL gotchas (pointer pre-advance; the force-vec guard since
|
||||
f16-GQA decode routes to tile), not on plumbing.
|
||||
3. **Vulkan last as the high-breadth capstone.** Reaches AMD + Intel + NVIDIA, but
|
||||
carries the most host glue and the coopmat2 limitation (NVIDIA GQA decode trades
|
||||
the fast path for scalar/cm1 only when a table is present). Do it once the
|
||||
pattern is proven on Metal + SYCL.
|
||||
|
||||
A cheaper variant (from the layer-2 doc, reaffirmed): ship **Metal + SYCL together**
|
||||
right after the ops PR and treat Vulkan as a separate later effort.
|
||||
|
||||
--------------------------------------------------------------------------------
|
||||
## 6. ROCm note
|
||||
|
||||
ROCm is in the **CUDA family**, not a separate port: patch 0030's allow-list already
|
||||
admits `"CUDA"/"ROCm"/"MUSA"`, and the CUDA kernels compile for HIP, so benefits #1
|
||||
and #2 are largely already-built or near-free on ROCm rather than a from-scratch
|
||||
accelerator port. Two caveats:
|
||||
|
||||
- **FP4-MMA (benefit #4) stays NVIDIA-Blackwell-only** - AMD has no native FP4-MMA,
|
||||
so the NVFP4 path dequants on ROCm exactly as elsewhere.
|
||||
- **The block-table read's force-vec routing matters on AMD too.** The AMD fast FA
|
||||
path is the wmma/mma kernel (`fattn-wmma-f16`), which - like CUDA mma, Metal MM
|
||||
and Vulkan cm2 - ignores the block table; the CUDA dispatch guard already forces a
|
||||
block-table op onto the vec kernel, so ROCm inherits correct routing, but the
|
||||
perf trade (vec vs wmma for AMD GQA decode with a table present) should be
|
||||
measured on AMD hardware before claiming a win. The GDN fusions, being plain
|
||||
CUDA-C, port to HIP with the rest of the CUDA path.
|
||||
|
||||
Net: ROCm is a "validate, don't re-port" follow-up - confirm the HIP build picks up
|
||||
the fusions + the force-vec block-table routing and gate it with `test-backend-ops`
|
||||
on an AMD GPU. It is genuinely separate from, and lighter than, the Metal / SYCL /
|
||||
Vulkan ports.
|
||||
|
||||
--------------------------------------------------------------------------------
|
||||
## 7. Summary
|
||||
|
||||
- **Benefit #3 (decode-first scheduler) is free and already portable** - host-side
|
||||
policy, zero kernel work; it only needs to be left enabled / upstreamed.
|
||||
- **Benefits #1 (GDN fusions) and #2 (block-table FA read) are the real ports** -
|
||||
both are vec/scalar-decode-kernel + `supports_op`/dispatch-guard changes, both
|
||||
rank Metal-then-SYCL-then-Vulkan, and they bundle into one per-backend PR behind a
|
||||
shared ops-first PR.
|
||||
- **Benefit #2 is the lever that makes paged KV non-negative off CUDA** - it removes
|
||||
the host-gather overhead the README measured as neutral-to-slightly-negative on
|
||||
the Mac. Feasibility: SYCL EASY, Metal EASY-MEDIUM, Vulkan MEDIUM. The universal
|
||||
constraint is that only the vec/scalar kernel admits the indexed read; the fast
|
||||
MM/coopmat2 path is contiguous-only, so route block-table ops onto vec (as CUDA
|
||||
already does) and leave the fast path's null-table read byte-identical.
|
||||
- **Benefit #4 (NVFP4 FP4-MMA) is out of scope** (Blackwell only); 0023's de-dup and
|
||||
0025's graph-coverage have backend-agnostic *ideas* but no lift-and-shift port.
|
||||
- **ROCm rides the CUDA path** (validate, don't re-port); FP4-MMA stays Blackwell-only.
|
||||
- Everything is bit-exact per-backend BY CONSTRUCTION (position-ordered table +
|
||||
address-only redirect = identical reduction order), gated by `test-backend-ops`
|
||||
(backendX-vs-CPU) **on the target hardware**, which we do not have here.
|
||||
</content>
|
||||
</invoke>
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,422 +0,0 @@
|
||||
# DECODE_SERVING_SCOPE - the continuous-serving decode gap
|
||||
|
||||
**Status: S1 + S3 IMPLEMENTED, GPU-validated, bit-exact, shipped as patches
|
||||
0040 (S1) + 0041 (S3). S2 DROPPED (measured non-target). See the results block
|
||||
below; the rest of this doc is the design/rationale those patches implement.**
|
||||
|
||||
## Results (GB10, measured)
|
||||
|
||||
Phase 0 confirmed host-bound: serving graph reuse **0% over ~5k steps** (layer-A
|
||||
rebuilds every step), `hostproc` 3.44 ms/step vs 1.59 static - the +1.85 ms IS the
|
||||
graph rebuild; `set_inputs` 0.047 ms and block-table 0.002 ms are negligible.
|
||||
|
||||
- **S1 (patch 0040)** - root cause: the paged decode inputs never overrode
|
||||
`can_reuse` (defaults false), so the graph could never be reused. Fixed with a
|
||||
256-bucketed-shape `can_reuse` + live-mctx refresh. Static batched-bench A/B:
|
||||
paged decode reuse **0% -> 95.5%**, bit-exact (md5 byte-identical reuse on/off).
|
||||
Necessary but **not** sufficient in serving (13.8% reuse alone - prefill
|
||||
co-batching churns the shape).
|
||||
- **S3 (patch 0041)** - keeps prefill out of decode steps so the scheduler emits
|
||||
reuse-stable pure-decode steps. **S1+S3 together (128-client staggered serving,
|
||||
MoE Qwen3.6-35B-A3B-NVFP4): reuse 0% -> 72.2%, `hostproc` 15.98 -> 6.31 ms/step,
|
||||
decode 4.05 -> 5.52 tok/s/seq median (4.24 -> 5.96 mean, at vLLM's ~5.9).**
|
||||
- **S2 (double-buffer set_inputs) - DROPPED.** Phase 0 put `set_inputs` at
|
||||
~0.05 ms/step: it is not the cost (the rebuild is), so S2 has nothing to recover.
|
||||
- **Follow-up to ~100% reuse - PADDED/FIXED-SLOT DECODE SHAPE: IMPLEMENTED,
|
||||
GPU-TESTED, REJECTED (not shipped).** See the "Padded-shape lever - rejected"
|
||||
block below. Summary: it does NOT close the serving gap. Padding holds the
|
||||
pure-decode width constant by emitting masked-inert dummy decodes for idle
|
||||
slots, and it is provably inert (single-seq md5 bit-exact + per-stream
|
||||
noise-floor determinism), but it **regresses throughput at every concurrency**
|
||||
(catastrophically at low load) because the serving decode here is
|
||||
**GPU-compute-bound, not host-rebuild-bound** - so the dummy-row compute it adds
|
||||
costs more than the graph-reuse it recovers. The original "remaining ~28% is
|
||||
request-boundary churn -> pad it" hypothesis stands mechanically, but the payoff
|
||||
premise (closing reuse pulls decode toward vLLM) is **not supported by
|
||||
measurement**.
|
||||
|
||||
---
|
||||
|
||||
## Padded-shape lever - rejected (implemented + GPU-tested, 2026-06-28)
|
||||
|
||||
The S1 section-(a) **padded / fixed-slot decode shape** was implemented in an
|
||||
isolated worktree off the committed S1/S3/tail base (paged HEAD `05eceb4`), built
|
||||
CUDA-only, and benched on GB10. **Verdict: REJECTED - it regresses serving
|
||||
throughput and does not close the vLLM gap.** Recorded here so it is not re-tried.
|
||||
|
||||
**Implementation** (default-off, `LLAMA_PAGED_PAD_DECODE=1`; `LLAMA_PAGED_PAD_WIDTH`
|
||||
caps the slot range): at the end of `pre_decode()`, on any step where no prompt
|
||||
tokens were admitted (`n_prompt_budgeted == 0`) and there is decode load, emit a
|
||||
masked-inert dummy decode for **every IDLE slot** (`batch.add(slot.id, 0,
|
||||
pos_max+1, /*output=*/true)`; cold slot -> fresh pos-0). This holds `n_tokens`,
|
||||
`n_seqs`, `n_seqs_unq`, `n_outputs` and the participating seq-id SET constant
|
||||
across arrivals/completions. A `release()`-side guard keeps a finished slot warm
|
||||
under padding (else patch 0024's reclaim-on-idle frees its KV and the next-step
|
||||
pos-0 re-warm churns paged-block allocation, destroying reuse). Each dummy is its
|
||||
OWN sequence, so its recurrent (gated-DeltaNet) state is private and its paged
|
||||
attention reads only its own cells; its logits are computed but never read
|
||||
(`post_decode()` only consumes `slot.i_batch` of GENERATING slots).
|
||||
|
||||
**Gates.** (1) Single-seq greedy md5 **bit-exact PASS** - dense
|
||||
`5951a5b4d624ce891e22ab5fca9bc439`, paged-MoE `8cb0ce23777bf55f92f63d0292c756b0`
|
||||
(the lever lives only in `llama-server`'s `update_slots()`, never in
|
||||
`llama-completion`). (2) **Per-stream serving determinism**: the literal
|
||||
"ON-vs-OFF token sequences identical" gate is **unachievable** - concurrent
|
||||
cuBLAS/FA decode is **not bit-reproducible run-to-run** even with padding OFF
|
||||
(OFF-vs-OFF diverging streams: dense 3/16, MoE 8/16, lockstep K=16). The
|
||||
**achievable inertness gate PASSED**: per-stream prefix-agreement ON-vs-OFF equals
|
||||
the OFF-vs-OFF noise floor exactly (MoE 0.940/0.940, dense 0.812/0.812), i.e. the
|
||||
dummy slots inject no systematic divergence beyond the pre-existing concurrent FP
|
||||
noise. So padding is provably inert; it just does not help.
|
||||
|
||||
**Bench (MoE Qwen3.6-35B-A3B-NVFP4, GB10).** Burst h2h, decode tok/s/seq:
|
||||
|
||||
| n | S1+S3 | PAD | vLLM |
|
||||
|-----|-------|------|------|
|
||||
| 8 | 28.16 | 6.05 | 44.8 |
|
||||
| 32 | 11.66 | 4.84 | 17.45|
|
||||
| 64 | 7.16 | 4.33 | 11.07|
|
||||
| 128 | 4.53 | 4.32 | 6.87 |
|
||||
|
||||
Staggered (`serve_bench.py` k=128 n=160 stagger0.25), aggregate decode tok/s and
|
||||
graph-reuse: baseline (reuse 0%) **757.6**; S1+S3 (reuse 72%) **763.3**; **PAD
|
||||
(reuse 38%) 558.0**.
|
||||
|
||||
**Why it fails (four independent reasons):**
|
||||
|
||||
1. **Serving decode is GPU-compute-bound, not host-rebuild-bound (this run).**
|
||||
Baseline reuse 0% (757.6 agg) is statistically equal to S1+S3 reuse 72% (763.3
|
||||
agg): `hostproc` is only ~4-8% of the per-step wall, so eliminating the host
|
||||
graph rebuild buys ~nothing. (This **corrects the host-bound hypothesis** above
|
||||
for this hardware: the earlier 542->762 host-bound delta did **not** reproduce
|
||||
- it was GPU-state/contention variance, not a stable reuse effect.)
|
||||
2. **Padding ADDS dummy-row compute** (full-width decode), costing throughput in
|
||||
direct proportion to `pad_width - real_load`: catastrophic at low concurrency
|
||||
(n=8: 28.16 -> 6.05, ~4.6x slower, because 8 real streams pay for a 128-wide
|
||||
step).
|
||||
3. **In continuous serving padding can't even hold the width constant**: arrivals
|
||||
are perpetually mid-prefill, so the idle-slot count varies and reuse DROPS
|
||||
72% -> 38% (the opposite of the goal). It only stabilises the pure-decode
|
||||
*tail* of a burst (verified: width pinned at 64 as real decoders fell 49->5),
|
||||
which is exactly where the dummy compute is most wasteful.
|
||||
4. **The completion-driven batch shrink that padding prevents is itself a
|
||||
throughput WIN** in a compute-bound regime (fewer real streams -> cheaper
|
||||
steps -> survivors finish faster); forcing constant width forfeits it.
|
||||
|
||||
**Conclusion.** The residual burst gap (paged 4.53 vs vLLM 6.87 at n=128 ~= 66%)
|
||||
is a **GPU-compute** gap (vLLM's MoE decode kernel + scheduler are ~1.3x faster on
|
||||
aggregate), not a host-loop gap. A host-side graph-reuse lever cannot close it.
|
||||
Do not re-pursue padded/fixed-slot shapes for throughput; if the host loop is ever
|
||||
re-confirmed dominant on other hardware (re-run reason 1's baseline-vs-S1+S3 A/B
|
||||
first), revisit - but only with an *adaptive* width matched to live load, never a
|
||||
fixed pad-to-`--parallel`.
|
||||
|
||||
---
|
||||
|
||||
Per the
|
||||
"profile-don't-assume" rule in
|
||||
[`.agents/vllm-parity-methodology.md`](../../../../.agents/vllm-parity-methodology.md),
|
||||
**Phase 0 (section 5) is to confirm the bottleneck on GPU before touching any
|
||||
code.** Everything below the Phase-0 line is a hypothesis ranked by
|
||||
value/effort/risk, not a measured result.
|
||||
|
||||
> **Regime warning (read first).** Every "decode is at the BW floor / ties vLLM"
|
||||
> and "host scheduling loop is the structural residual" conclusion in
|
||||
> [`README.md`](../README.md) section 5 was measured with **`llama-batched-bench`**:
|
||||
> a STATIC serving width (fixed `npl`, all sequences in lockstep, constant
|
||||
> batch shape every step). That is the **decode KERNEL** regime, and there the
|
||||
> patch series is at parity (paged ~6.1 tok/s/seq vs vLLM ~5.9 at npl128). This
|
||||
> document is about a **different regime**: real **continuous SERVING** through
|
||||
> `llama-server`'s `update_slots()` loop, where requests arrive and complete
|
||||
> asynchronously, the batch shape churns every step, and paged drops to ~3.7
|
||||
> tok/s/seq (-39%) while vLLM sustains ~5.9. The gap is the **scheduler / host
|
||||
> loop**, not the kernel. This is the serving analogue of the prefill-GEMM regime
|
||||
> split called out in [`PREFILL_GEMM_SCOPE.md`](PREFILL_GEMM_SCOPE.md).
|
||||
|
||||
Cross-links: [`README.md`](../README.md) sections 2 (scheduler), 3 (patches
|
||||
0008/0013/0016/0024/0025/0029), 5 (rejected levers - lever 2 graph coverage was
|
||||
FLAT *in the static regime*; this doc reopens it for the *serving* regime);
|
||||
[`.agents/llama-cpp-localai-paged-backend.md`](../../../../.agents/llama-cpp-localai-paged-backend.md)
|
||||
(bit-exact gate);
|
||||
[`.agents/vllm-parity-methodology.md`](../../../../.agents/vllm-parity-methodology.md)
|
||||
(both-engine ground-truth, per-lever A/B, record-rejected-levers).
|
||||
|
||||
---
|
||||
|
||||
## 1. The two regimes, and why the kernel-parity result does not carry over
|
||||
|
||||
`llama-batched-bench` and a real serving workload exercise the **same decode
|
||||
kernels** but **different host loops**:
|
||||
|
||||
| | `llama-batched-bench` (kernel regime) | `llama-server` continuous serving |
|
||||
|---|---|---|
|
||||
| batch shape per step | **constant** (fixed `npl`, lockstep) | **churns** (arrivals/completions, interleaved prefill) |
|
||||
| participating seq-set | **fixed** for the whole run | **changes** as requests start/finish |
|
||||
| graph reuse (see s.2) | holds after warmup -> 1 capture, replayed | breaks nearly every step -> rebuild + re-capture |
|
||||
| measured | paged ~6.1 tok/s/seq ~ vLLM ~5.9 | paged ~3.7 vs vLLM ~5.9 (-39%) |
|
||||
|
||||
The README's decode parity, BW-floor, and "host loop is the irreducible
|
||||
residual" findings are all **kernel-regime** findings. They prove the *kernels*
|
||||
are not the serving gap. They do **not** prove the host loop is irreducible *in
|
||||
serving* - the static bench holds the batch shape constant, which is exactly the
|
||||
condition that lets both graph-reuse layers (section 2) stay hot. Serving
|
||||
violates that condition. So the serving gap is reopened here as a host /
|
||||
scheduler problem, orthogonal to the kernel.
|
||||
|
||||
---
|
||||
|
||||
## 2. Root-cause hypothesis (from source, pin `9d5d882d` + the dev tree)
|
||||
|
||||
There are **two independent graph-reuse layers**, and continuous batching breaks
|
||||
**both** on nearly every step. This is the leading hypothesis for the -39%.
|
||||
|
||||
### 2a. Layer A - llama-context graph reuse (`can_reuse` / `allow_reuse`)
|
||||
|
||||
`llama_context::process_ubatch` (`src/llama-context.cpp` ~L1366) only **reuses
|
||||
the built ggml graph** when `res->can_reuse(gparams)` holds. `allow_reuse`
|
||||
(`src/llama-graph.h` ~L631) requires, among others:
|
||||
|
||||
```
|
||||
ubatch.n_tokens == other.ubatch.n_tokens &&
|
||||
ubatch.n_seqs == other.ubatch.n_seqs &&
|
||||
ubatch.n_seqs_unq == other.ubatch.n_seqs_unq &&
|
||||
ubatch.equal_seqs() == other.ubatch.equal_seqs()
|
||||
// + (when equal_seqs) the participating sequence-id SET must match
|
||||
```
|
||||
|
||||
In serving, `n_tokens` changes whenever the decode load `D` changes or a prefill
|
||||
chunk is co-batched, and the **sequence-id set** changes whenever a request
|
||||
starts or finishes. Either makes `can_reuse` return false, so `process_ubatch`
|
||||
falls into the `else` branch: **rebuild the graph** (`model.build_graph`) +
|
||||
`ggml_backend_sched_reset` + `ggml_backend_sched_alloc_graph` - full host-side
|
||||
graph construction + allocation, **every step**. In batched-bench all sequences
|
||||
are lockstep so `n_tokens`/seq-set are constant and `can_reuse` is true after
|
||||
warmup (the `graphs reused = N` perf line is ~all steps).
|
||||
|
||||
### 2b. Layer B - CUDA graph capture (`ggml_cuda_graph_*`)
|
||||
|
||||
Even when layer A reuses, the CUDA backend re-checks
|
||||
`ggml_cuda_graph_update_required` (`ggml-cuda.cu` ~L3367): it `memcmp`s every
|
||||
node's `ne`, `nb`, and `src[]->data` pointers against the captured graph. Any
|
||||
shape change -> `cudaGraphExecUpdate` / re-instantiate. Two serving-specific
|
||||
triggers:
|
||||
|
||||
- **shape churn** (same root cause as layer A): different `n_tokens` -> different
|
||||
node `ne` -> update required.
|
||||
- **paged data-pointer churn**: when a co-batched prefill allocates new KV blocks
|
||||
(or a finished sequence frees them), the per-step KV view tensors' `data`
|
||||
pointers move, so even a constant-shape decode step can trip the `memcmp`. (The
|
||||
block-table *contents* live in a fixed device buffer filled by `set_inputs`, so
|
||||
the table tensor pointer itself is stable - 0029 keeps that cheap - but the K/V
|
||||
cache views are not.)
|
||||
|
||||
Net: under serving, the GPU sits idle between launches while the host rebuilds
|
||||
the graph (layer A) and re-instantiates the CUDA graph (layer B), then runs an
|
||||
un-graphed `set_inputs` (H2D input copies) before each launch. vLLM avoids this
|
||||
with **padded/bucketed decode batch shapes + piecewise CUDA graphs**: it pads the
|
||||
decode batch to a fixed set of sizes and captures one persistent graph per
|
||||
bucket, so the steady-state decode step is a single `cudaGraphLaunch` with no
|
||||
host rebuild. Its scheduler is also a tight C++ loop with chunked-prefill
|
||||
interleave that keeps the GPU fed.
|
||||
|
||||
### 2c. Per-step host work that runs un-graphed regardless (already instrumented)
|
||||
|
||||
The dev tree carries a built-in `[L5INSTR]` profiler (`src/paged-attn.cpp`,
|
||||
hooks in `src/llama-context.cpp` and `src/llama-kv-cache.cpp`) that already
|
||||
isolates the host buckets we care about, printed at process exit:
|
||||
|
||||
```
|
||||
[L5INSTR] get_block_table n=.. sum=..ms mean=..ms | set_inputs n=.. mean=..ms | hostproc n=.. mean=..ms
|
||||
```
|
||||
|
||||
- `hostproc` = `mctx->apply()` + graph reuse-check/rebuild + `set_inputs`, i.e.
|
||||
the whole host window **before** `graph_compute` (it does NOT include the GPU
|
||||
launch). Prior profiles put this near ~1.4 ms/step.
|
||||
- `set_inputs` = the H2D input fills (positions, masks, block table, idxs).
|
||||
- `get_block_table` = the paged block-table host build (0029 caches it
|
||||
within-step; `LLAMA_PAGED_NO_BT_CACHE` A/B-toggles that).
|
||||
|
||||
If `hostproc` per step is a large fraction of the serving per-step wall time
|
||||
(and the `graphs reused` count is low), the gap is host-bound, not kernel-bound.
|
||||
|
||||
### 2d. The serial-SSM host loop (named in README s.5, secondary here)
|
||||
|
||||
The gated-DeltaNet decode advances recurrent state per step; sampling cannot
|
||||
start until logits land. The README already names this as a structural floor in
|
||||
the *kernel* regime. It is the same in serving but is the *smaller* term - the
|
||||
graph-rebuild/re-capture overhead (2a/2b) is the new, serving-specific cost the
|
||||
static bench hides, and it is the one to attack first.
|
||||
|
||||
---
|
||||
|
||||
## 3. What the already-shipped scheduler patches do (and do NOT do)
|
||||
|
||||
These exist; understand them before proposing anything. **None of them touch the
|
||||
two graph-reuse layers** - they target prefill freezing and burst collapse, not
|
||||
steady-state decode-step host overhead. That is why the serving gap survives them.
|
||||
|
||||
| Patch | What it does | What it does NOT do |
|
||||
|---|---|---|
|
||||
| 0008 cross-request prefix-share (server loop) | Concurrent shared-prefix requests prefill only the divergent suffix (fewer prefill tokens). | Does not stabilise decode batch shape; does not graph-reuse. |
|
||||
| 0013 `LLAMA_PREFILL_BUDGET` | Static per-step prefill-token cap (vLLM `--max-num-batched-tokens` analogue); flattens the ITL spike a long prefill inflicts on co-batched decode. | Ignores decode load; per-workload tuning; no effect on decode-step graph reuse. |
|
||||
| 0016 dynamic decode-first budget | `max(n_ubatch, T-D)` leftover-after-decode budget + per-slot chunk cap; decode claimed first, auto-shrinks as `D` rises. Stops a prefill chunk from inflating the step past `T`. | **Still lets the per-step decode `n_tokens` and seq-set vary**, so it does not make the decode step graph-reusable; it shapes prefill admission, not decode-shape stability. |
|
||||
| 0024 paged-pool burst-reclaim | Truncate/defrag/release KV blocks; fixes long-server prefill burst collapse (488->44->532 t/s). | Host accounting only; nothing about decode-step graph capture. |
|
||||
| 0025 `LLAMA_MOE_FORCE_GRAPHS` | Keeps CUDA graphs ON for the grouped-MMQ MoE decode step (lifts the conservative `MUL_MAT_ID` graph-disable). | Helps the CUDA-graph *eligibility* of one op; does **not** make layer-A/B *reuse* hold across churning steps. It is necessary-not-sufficient: a step that rebuilds anyway gets recaptured regardless. |
|
||||
| 0029 block-table within-step cache | `get_block_table` computed once per step, memcpy'd to other full-attn layers (-87/-91%). | Shrinks one `set_inputs`/`hostproc` sub-term; does not address rebuild/re-capture. |
|
||||
|
||||
**README s.5 "lever 2 (graph/stream coverage): FLAT"** was concluded **in the
|
||||
static batched-bench regime**, where graphs already reuse - so more graph
|
||||
coverage was correctly a no-op there. That conclusion does **not** apply to the
|
||||
serving regime, where graphs do **not** reuse. This doc reopens graph coverage
|
||||
**for serving only**; record it as a regime-scoped reopening, not a contradiction.
|
||||
|
||||
---
|
||||
|
||||
## 4. Ranked lever plan (hypotheses - gate on Phase 0 first)
|
||||
|
||||
Ranked by value/effort with bit-exactness/risk called out. All are **host-side /
|
||||
scheduler** levers (no decode-kernel changes), so all are *bit-exact-safe by
|
||||
construction* provided padding tokens are masked-inert and verified against the
|
||||
per-path md5 gate.
|
||||
|
||||
### Lever S1 (TOP) - bucketed/padded decode-step shape for graph reuse
|
||||
|
||||
**Value: high (targets the dominant -39% mechanism). Effort: medium-high. Risk:
|
||||
medium (correctness of padding inertness; seq-set churn is harder than n_tokens).**
|
||||
|
||||
Make the steady-state decode step present a **stable, bucketed shape** to both
|
||||
reuse layers, mirroring vLLM's padded decode batch + piecewise CUDA graphs:
|
||||
|
||||
- Pad the per-step decode `n_tokens` (and the stream/seq count the graph sees) up
|
||||
to the next bucket in a small fixed set (e.g. {power-of-two or fixed grid}), so
|
||||
`allow_reuse` (layer A) and `update_required` (layer B) hold across steps with
|
||||
the same bucket. Padding tokens are dummy, masked positions that contribute
|
||||
nothing to any real sequence's logits.
|
||||
- Bound the number of distinct live buckets so a handful of persistent CUDA
|
||||
graphs cover steady decode (vLLM captures ~tens).
|
||||
- Handle the seq-set component of `allow_reuse`: bucketing `n_tokens` alone is
|
||||
insufficient because the *participating sequence-id set* must also match. Either
|
||||
(a) pad to a fixed stream-slot layout so the seq-set is stable across arrivals
|
||||
/completions, or (b) relax/extend the reuse key so a pure-decode step keyed on
|
||||
bucket+slot-layout reuses regardless of which slots are occupied. (b) is the
|
||||
higher-leverage but more invasive option.
|
||||
|
||||
Bit-exact gate: greedy md5 per path with padding ON must equal the recorded
|
||||
references (`5951a5b4` dense, `8cb0ce23` paged-MoE); `test-backend-ops`
|
||||
unaffected (no op changes). The risk is that masked/padded positions leak into a
|
||||
real logit (off-by-one in the mask) - the md5 gate catches it.
|
||||
|
||||
### Lever S2 - overlap per-step host work with GPU decode (double-buffer inputs)
|
||||
|
||||
**Value: medium-high (recovers the `hostproc` window even when S1 partial).
|
||||
Effort: medium. Risk: low (host-side reordering only, bit-exact-safe).**
|
||||
|
||||
Even with graphs reused, `set_inputs` (+ the pre-`set_inputs` sync) runs
|
||||
un-graphed and serially *before* each launch (`hostproc` ~1.4 ms/step in prior
|
||||
profiles). Overlap the host scheduling + input build of step N+1 with the GPU
|
||||
decode of step N: double-buffer the input device tensors so the host can fill
|
||||
N+1's inputs while N's graph is in flight, and prepare the next ubatch / block
|
||||
table on the host concurrently. This is the llama.cpp analogue of vLLM keeping
|
||||
the GPU fed. Strictly host-side, no numeric change -> bit-exact. (0029 already
|
||||
banks part of this for the block table within a step; S2 extends it across
|
||||
steps.)
|
||||
|
||||
### Lever S3 - graph-shape-stable scheduling (bridge from 0016)
|
||||
|
||||
**Value: medium (multiplies S1; low marginal value without S1). Effort: low-medium
|
||||
(extends the existing 0016 policy). Risk: low (scheduler policy, bit-exact when
|
||||
the decode result is unchanged).**
|
||||
|
||||
Extend the existing decode-first budget (0016) so the scheduler actively *prefers
|
||||
graph-reusable steps*: keep prefill chunks out of the decode step (run prefill in
|
||||
its own steps, or at a fixed chunk size) so the decode batch shape stays on a
|
||||
bucket rather than being perturbed by interleaved prefill tokens every step. This
|
||||
is the policy half of S1 - S1 makes a bucketed step reusable; S3 makes the
|
||||
scheduler emit bucketed steps. Pair them.
|
||||
|
||||
**Rejected/deferred (record so they are not re-tried):**
|
||||
|
||||
- **More CUDA-graph *coverage* alone (the README lever-2 redo): still FLAT
|
||||
without S1.** Forcing more ops graph-eligible (beyond 0025) does nothing while
|
||||
layer A rebuilds the graph every step - the recapture dominates. Only valuable
|
||||
*after* S1 makes reuse hold.
|
||||
- **`GGML_CUDA_DISABLE_GRAPHS` / disabling graphs in serving: REJECTED a priori
|
||||
as a fix** (it is an A/B *probe* for Phase 0, not a lever) - it removes capture
|
||||
cost but also removes replay benefit; expected net-negative.
|
||||
- **Precision levers (W4A16, bf16-SSM): out of scope** - this gap is host-bound,
|
||||
not GEMM/BW-bound (see README s.5 rejections; do not reopen).
|
||||
|
||||
---
|
||||
|
||||
## 5. Phase 0 - confirm it is host-bound BEFORE building (run when the GPU frees)
|
||||
|
||||
Do NOT build any lever until this confirms host-bound. The dev tree already has
|
||||
all the instrumentation; this is a measurement, not a code change. **One GPU
|
||||
bencher at a time** (GPU-contention rule).
|
||||
|
||||
**Workload.** Real continuous serving, not batched-bench: run `llama-server`
|
||||
(paged build) with the paged config and drive it with a steady concurrent
|
||||
streaming load (e.g. a K-client async generator hitting `/completion` with
|
||||
staggered arrivals so requests start/finish asynchronously - the regime
|
||||
batched-bench cannot produce). Use the same models/flags as README s.4:
|
||||
`-fa on -ngl 99`, `LLAMA_KV_PAGED=1` (+ `LLAMA_MOE_FORCE_GRAPHS=1` for MoE),
|
||||
dense Qwen3.6-27B-NVFP4 and MoE Qwen3.6-35B-A3B-NVFP4. Pick K so the *effective
|
||||
decode width* matches a static `npl` you have a kernel-regime number for (e.g.
|
||||
~128) - that gives the apples comparison: static 6.1 vs serving 3.7 tok/s/seq.
|
||||
|
||||
**Signals to capture (all already exist):**
|
||||
|
||||
1. **Graph reuse rate.** The `graphs reused = N` perf line (`llama-context.cpp`
|
||||
~L4146, from `data.n_reused`) over total decode steps. Hypothesis: ~100% in
|
||||
batched-bench, near 0% in serving. This is the single most decisive number.
|
||||
A/B with `LLAMA_GRAPH_REUSE_DISABLE=1` (forces the rebuild path) - if serving
|
||||
is already near that floor, layer-A reuse is the gap.
|
||||
2. **`[L5INSTR]` host buckets** (printed at exit): `hostproc`, `set_inputs`,
|
||||
`get_block_table` mean ms/step. Compare serving vs batched-bench. A/B the
|
||||
block-table cache with `LLAMA_PAGED_NO_BT_CACHE`.
|
||||
3. **GPU-busy %** in a steady-state serving window via nsys (sum of kernel
|
||||
durations / wall) and the **inter-launch host gap** (time between consecutive
|
||||
`cudaGraphLaunch`/kernel launches). Hypothesis: batched-bench ~96-99% busy
|
||||
(README/methodology note the early "low util" was a window artifact); serving
|
||||
materially lower, with the gap ~= `hostproc`/step. *Watch the same window
|
||||
artifact* the methodology warns about - measure a clean steady-state span.
|
||||
4. **CUDA-graph re-instantiation count** - confirm layer B is also re-capturing
|
||||
(nsys shows `cudaGraphInstantiate`/`cudaGraphExecUpdate` per step, or add a
|
||||
host-side counter print - host-side only, no kernel code).
|
||||
|
||||
**Decision rule.** Host-bound (proceed with S1/S2/S3) if: serving `graphs reused`
|
||||
is low AND `hostproc`/step is a large fraction of serving per-step wall AND
|
||||
GPU-busy% drops vs batched-bench by ~the observed throughput ratio (~3.7/6.1).
|
||||
If instead GPU-busy% stays high and per-kernel time grows, the cause is
|
||||
elsewhere (e.g. serving runs a worse effective batch shape into the kernels) -
|
||||
re-scope before building.
|
||||
|
||||
**Ground-truth vLLM (both-engine rule).** Capture vLLM at the same concurrency:
|
||||
GPU-busy% / step cadence (nsys) and its scheduler step time. Confirm vLLM stays
|
||||
GPU-bound (persistent graphs) where paged goes host-bound - that is the
|
||||
direct evidence the gap is the host loop, and it sizes the achievable win.
|
||||
|
||||
---
|
||||
|
||||
## 6. Summary
|
||||
|
||||
- The serving gap (paged 3.7 vs vLLM 5.9 tok/s/seq, -39%) is a **host/scheduler**
|
||||
problem, distinct from the decode **kernel** (at parity in batched-bench). The
|
||||
README's BW-floor/host-loop-residual findings are kernel-regime and do not
|
||||
bound the serving regime.
|
||||
- Leading mechanism: continuous batching's **batch-shape + seq-set churn breaks
|
||||
both graph-reuse layers** (llama-context `can_reuse`, CUDA `update_required`)
|
||||
every step, so the GPU idles while the host rebuilds + re-captures + runs
|
||||
un-graphed `set_inputs`. vLLM avoids this with padded/bucketed decode shapes +
|
||||
piecewise CUDA graphs.
|
||||
- The shipped scheduler patches (0008/0013/0016/0024/0025/0029) target prefill
|
||||
freezing + burst collapse, **not** decode-step graph reuse - which is why the
|
||||
serving gap survives them.
|
||||
- Top levers (all host-side, bit-exact-safe): **S1** bucketed/padded decode-step
|
||||
shape for graph reuse, **S2** double-buffer/overlap per-step host work, **S3**
|
||||
graph-shape-stable scheduling (extend 0016). Gate everything on **Phase 0**:
|
||||
the `graphs reused` rate + `[L5INSTR]` host buckets + nsys GPU-busy% in real
|
||||
`llama-server` serving vs batched-bench, with vLLM ground-truthed at the same
|
||||
concurrency.
|
||||
</content>
|
||||
</invoke>
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -1,376 +0,0 @@
|
||||
# GB10 vLLM Parity Reopen Spec
|
||||
|
||||
Status: scoped follow-up. This document intentionally challenges the current
|
||||
`VLLM_PARITY_FINAL.md` conclusion that GB10 parity is closed. The final record is
|
||||
still useful as a baseline, but the follow-up work must treat it as a hypothesis
|
||||
to test, not as a proof of impossibility.
|
||||
|
||||
## Goal
|
||||
|
||||
Determine whether llama.cpp / ggml can close the remaining GB10 parity gap for
|
||||
Qwen3.6 NVFP4 hybrid gated-DeltaNet models by porting or adapting concrete vLLM
|
||||
implementation ideas, while preserving LocalAI's hard correctness gates.
|
||||
|
||||
Success means one of two outcomes:
|
||||
|
||||
1. A measured, source-backed path improves paged llama.cpp materially toward vLLM
|
||||
parity on GB10.
|
||||
2. The remaining gap is rejected with clean provenance: clean source, clean DGX
|
||||
host state, artifact-pinned A/B results, and explicit correctness gates.
|
||||
|
||||
## Non-goals
|
||||
|
||||
- Do not accept a "closed" conclusion based only on existing docs.
|
||||
- Do not run long builds or benchmarks without a recorded DGX preflight.
|
||||
- Do not edit `patches/paged/*.patch` directly. Kernel changes land fork-first in
|
||||
`mudler/llama.cpp:localai-paged`, then the LocalAI patch series is regenerated.
|
||||
- Do not treat a standalone PoC as a result. Every performance claim requires an
|
||||
in-backend A/B.
|
||||
- Do not ship lossy paths default-on. Non-byte-identical paths require KL gates.
|
||||
|
||||
## Required Preflight
|
||||
|
||||
Before any DGX build, benchmark, or profile:
|
||||
|
||||
1. `docker ps` must show no running containers, especially no `local-ai-worker`.
|
||||
2. `nvidia-smi --query-compute-apps=pid` must show zero compute apps.
|
||||
3. `~/gpu_bench_lock/owner` must be absent or `FREE*`.
|
||||
4. Record hostname, git SHA, dirty status, build arch, binary mtimes, model paths,
|
||||
benchmark command, and environment variables.
|
||||
|
||||
Use `~/_git/llama.cpp` as the local source of truth. DGX source trees are allowed
|
||||
for builds and artifact inspection, but dirty DGX checkouts must not be treated as
|
||||
canonical source.
|
||||
|
||||
## Evidence From Subagent Audit
|
||||
|
||||
Four read-only subagents audited the current state:
|
||||
|
||||
- llama.cpp / ggml source audit.
|
||||
- vLLM source and installed package audit.
|
||||
- LocalAI patch and docs audit.
|
||||
- DGX artifact and profile audit.
|
||||
|
||||
Their shared conclusion: the final docs are a useful snapshot, but several
|
||||
claims are broader than the available evidence.
|
||||
|
||||
Key findings:
|
||||
|
||||
- The strongest unresolved implementation target is W4A16 grouped MoE prefill.
|
||||
vLLM uses Marlin W4A16 on GB10, and llama.cpp already has a correct but untuned
|
||||
scaffold in `ggml/src/ggml-cuda/w4a16-gemm.cu`.
|
||||
- The existing W4A16 rejection is a first-implementation failure, not a proof of
|
||||
impossibility. The patch header names fixable costs: f32 to bf16 cast pre-pass,
|
||||
host tile-map setup, small copies, scalar dequant, and ragged tile waste.
|
||||
- The 924 t/s paged GPU-steady decode figure is artifact-backed, but the vLLM
|
||||
1078 t/s true GPU-steady figure was not found as a self-contained
|
||||
ntg16/ntg64 difference-method artifact. Reproduce before relying on the 86%
|
||||
claim.
|
||||
- GDN M5 is real, but M5/M8 provenance is muddy because CDEF records a dirty
|
||||
dev-tree M8 commit while docs describe production M5 defaults.
|
||||
- S3 fixed-period scheduling and fixed-slot padding were rejected, but adaptive
|
||||
scheduling remains unproven.
|
||||
|
||||
## Candidate Workstreams
|
||||
|
||||
### A. Provenance And Baseline Reproduction
|
||||
|
||||
Purpose: make later claims defensible.
|
||||
|
||||
Tasks:
|
||||
|
||||
- Build from clean `~/_git/llama.cpp` `localai-paged` source, or a clean DGX clone
|
||||
generated from that source.
|
||||
- Re-run canonical md5 gates for paged MoE and dense:
|
||||
- paged MoE: `8cb0ce23777bf55f92f63d0292c756b0`
|
||||
- dense: `5951a5b4d624ce891e22ab5fca9bc439`
|
||||
- Re-run a short prefill baseline for MoE and dense at `npp=512,2048`.
|
||||
- Re-run graph-node-traced decode for paged and vLLM using the same
|
||||
difference-method shape: `ntg=16` and `ntg=64` at N=128 or N=256.
|
||||
|
||||
Gate:
|
||||
|
||||
- No implementation work starts until the baseline artifact names, source SHAs,
|
||||
and commands are recorded.
|
||||
|
||||
### B. W4A16 Grouped MoE Prefill Attack
|
||||
|
||||
Purpose: port the vLLM Marlin W4A16 advantage into ggml's in-backend MoE prefill
|
||||
path.
|
||||
|
||||
Current hooks:
|
||||
|
||||
- `ggml/src/ggml-cuda/w4a16-gemm.cu`
|
||||
- `ggml/src/ggml-cuda/w4a16-gemm.cuh`
|
||||
- `ggml/src/ggml-cuda/ggml-cuda.cu` around `ggml_cuda_mul_mat_id`
|
||||
- `ggml/src/ggml-cuda/mmq.cu` around `LLAMA_W4A16_PREFILL_M`
|
||||
|
||||
Known current costs:
|
||||
|
||||
- Separate f32 to bf16 activation cast pass.
|
||||
- Host-built tile metadata and H2D copies.
|
||||
- Scalar in-register FP4 to bf16 dequant.
|
||||
- 4-byte weight staging.
|
||||
- Ragged expert tile waste.
|
||||
- Interaction with the generic token-sorting fallback.
|
||||
|
||||
Phased experiments:
|
||||
|
||||
1. Reconfirm current 0035 W4A16 performance with clean provenance.
|
||||
2. Remove or fuse the f32 to bf16 activation cast pre-pass.
|
||||
3. Move tile metadata generation device-side or cache it across repeated shapes.
|
||||
4. Improve weight staging width and shared-memory layout.
|
||||
5. Tune tile shapes for ragged per-expert M distribution.
|
||||
6. Compare against FP4-MMQ and vLLM Marlin buckets with nsys.
|
||||
|
||||
Correctness gate:
|
||||
|
||||
- `test-backend-ops MUL_MAT_ID` forced W4A16.
|
||||
- Greedy md5 for unaffected default-off path.
|
||||
- KL gate for engaged W4A16 path: `KLD(W4A16||f16) <= KLD(FP4-MMQ||f16)`.
|
||||
- Decode path unchanged when `LLAMA_W4A16_PREFILL_M=0`.
|
||||
|
||||
Benchmark gate:
|
||||
|
||||
- Beat default FP4-MMQ on MoE `S_PP` at `npp=512` and `npp=2048`.
|
||||
- No material peak-memory increase.
|
||||
- No decode regression in the default path.
|
||||
|
||||
### C. Native Ragged Grouped FP4-MMA Prefill
|
||||
|
||||
Purpose: test whether patch 0034's native FP4-MMA PoC failed due to integration,
|
||||
not due to the core kernel idea.
|
||||
|
||||
Current hooks:
|
||||
|
||||
- `ggml/src/ggml-cuda/fp4-gemm.cu`
|
||||
- `ggml/src/ggml-cuda/fp4-gemm.cuh`
|
||||
- `LLAMA_FP4_PREFILL_M`
|
||||
|
||||
Experiment:
|
||||
|
||||
- Build a graph-safe ragged grouped FP4-MMA MoE prefill kernel that avoids the
|
||||
per-expert host-sync loop.
|
||||
|
||||
Correctness gate:
|
||||
|
||||
- Same KL and op gates as W4A16.
|
||||
- Explicit proof that the per-expert host fallback is not on the hot path.
|
||||
|
||||
Benchmark gate:
|
||||
|
||||
- Beat current FP4-MMQ or lose decisively enough to close this branch.
|
||||
|
||||
### D. GDN Chunked Scan Follow-up
|
||||
|
||||
Purpose: compare vLLM's in-tree FLA-derived GDN path against the current M5
|
||||
implementation without relying on muddy dev-tree artifacts.
|
||||
|
||||
Current hooks:
|
||||
|
||||
- `ggml/src/ggml-cuda/gated_delta_net.cu`
|
||||
- `GDN_TC`
|
||||
- `GDN_CHUNK_MIN`
|
||||
- existing M5 tensor-core ladder
|
||||
|
||||
Phased experiments:
|
||||
|
||||
1. Clean A/B: current production M5 against sequential and against recorded M8
|
||||
dev-tree behavior.
|
||||
2. C=32 and C=64 variants.
|
||||
3. dv slab variants.
|
||||
4. cp.async staging variants.
|
||||
5. Register-state variant only if the lower-risk variants show headroom.
|
||||
|
||||
Correctness gate:
|
||||
|
||||
- `test-backend-ops GATED_DELTA_NET`, including multi-chunk, tail-chunk,
|
||||
multi-seq, and adversarial decay cases.
|
||||
- Greedy md5 per path.
|
||||
- KL gate for any non-byte-identical path.
|
||||
|
||||
Benchmark gate:
|
||||
|
||||
- Beat current M5, not just old sequential.
|
||||
- Preserve decode behavior by keeping `GDN_CHUNK_MIN > 1`.
|
||||
|
||||
### E. MoE Weighted Fan-in Fusion
|
||||
|
||||
Purpose: remove generic graph-level MoE reduction overhead that vLLM avoids or
|
||||
amortizes through fused MoE handling.
|
||||
|
||||
Current source:
|
||||
|
||||
- `src/llama-graph.cpp`, MoE down projection and expert reduction.
|
||||
- `ggml/src/ggml-cuda/ggml-cuda.cu`, CUDA fusion and MoE support.
|
||||
|
||||
Experiment:
|
||||
|
||||
- Add a CUDA-specific fused path for `down_experts * weights -> sum expert_used`
|
||||
while preserving the current reduction order where required.
|
||||
|
||||
Correctness gate:
|
||||
|
||||
- Bit-exact for supported shapes, or KL-benign if reduction order changes.
|
||||
- Handles all `n_expert_used` used by Qwen3.6 MoE.
|
||||
|
||||
Benchmark gate:
|
||||
|
||||
- Move MoE prefill or decode wall time by more than noise. If it is only a
|
||||
2-3% dispatch bucket, record and deprioritize.
|
||||
|
||||
### F. Adaptive Serving Scheduler
|
||||
|
||||
Purpose: keep S3's decode-window benefit without reproducing its TTFT collapse.
|
||||
|
||||
Current hooks:
|
||||
|
||||
- `tools/server/server-context.cpp`
|
||||
- `LLAMA_PAGED_DECODE_STABLE`
|
||||
- `LLAMA_PAGED_PREFILL_PERIOD`
|
||||
- existing dynamic prefill budget patches.
|
||||
|
||||
Experiment:
|
||||
|
||||
- Replace fixed-period prefill deferral with adaptive admission based on live
|
||||
decode width, waiting prefill backlog, and TTFT budget.
|
||||
|
||||
Correctness gate:
|
||||
|
||||
- Serving output correctness unchanged.
|
||||
- No starvation of prefill requests.
|
||||
|
||||
Benchmark gate:
|
||||
|
||||
- Improve aggregate throughput or decode throughput at N=128 or N=256 without
|
||||
the 2.5x TTFT regression from fixed S3.
|
||||
|
||||
### G. Projection And GDN Glue Fusion
|
||||
|
||||
Purpose: steal vLLM's `prepare_gdn_attention_core_inputs` idea where ggml still
|
||||
pays small copy, cat, slice, or unpack kernels.
|
||||
|
||||
Current source:
|
||||
|
||||
- `src/models/qwen35.cpp`
|
||||
- `src/models/qwen35moe.cpp`
|
||||
- `ggml/src/ggml-cuda/ggml-cuda.cu`
|
||||
|
||||
Experiment:
|
||||
|
||||
- Fuse q/k/v/z unpacking, BA projection preparation, RMSNorm-gated output prep,
|
||||
and FP4/FP8 quant prep where the graph pattern is stable.
|
||||
|
||||
Correctness gate:
|
||||
|
||||
- Per-op tests for new fusion.
|
||||
- Greedy md5 for model paths.
|
||||
|
||||
Benchmark gate:
|
||||
|
||||
- Only continue if nsys shows this bucket is material after MoE and GDN work.
|
||||
|
||||
## Subagent Plan
|
||||
|
||||
Use subagents for independent read, implementation, and review slices. Do not use
|
||||
subagents to edit the same files in parallel.
|
||||
|
||||
Recommended roles by phase:
|
||||
|
||||
- Phase 0 source/provenance agent: owns command capture and source SHA checks.
|
||||
- Phase 0 artifact agent: owns parsing existing and new benchmark artifacts.
|
||||
- W4A16 kernel agent: owns `w4a16-gemm.*`.
|
||||
- W4A16 integration agent: owns `ggml-cuda.cu` and `mmq.cu` dispatch plumbing.
|
||||
- GDN kernel agent: owns `gated_delta_net.cu`.
|
||||
- Scheduler agent: owns server scheduling files only.
|
||||
- Reviewer agent: reviews gates, provenance, and whether measured claims match
|
||||
artifacts.
|
||||
|
||||
Subagent output requirements:
|
||||
|
||||
- File paths and functions inspected or changed.
|
||||
- Exact commands run.
|
||||
- Exact artifacts produced.
|
||||
- Pass/fail result against the phase gate.
|
||||
- Any uncertainty labeled explicitly.
|
||||
|
||||
## Phase Order
|
||||
|
||||
### Phase 0 - Reproduce And Correct The Record
|
||||
|
||||
Do first.
|
||||
|
||||
Deliverables:
|
||||
|
||||
- Clean source/build provenance.
|
||||
- Short prefill baseline.
|
||||
- Graph-node-traced decode difference-method for paged and vLLM.
|
||||
- Updated docs if the 86% decode claim or CDEF provenance changes.
|
||||
|
||||
Exit criteria:
|
||||
|
||||
- Baseline is trustworthy enough to judge optimization deltas.
|
||||
|
||||
### Phase 1 - W4A16 MoE Prefill
|
||||
|
||||
Do second.
|
||||
|
||||
Deliverables:
|
||||
|
||||
- Reconfirmed current W4A16 baseline.
|
||||
- At least one targeted W4A16 overhead removal.
|
||||
- A/B against default FP4-MMQ.
|
||||
|
||||
Exit criteria:
|
||||
|
||||
- Either W4A16 beats FP4-MMQ and continues, or it is rejected with direct
|
||||
artifact-backed evidence.
|
||||
|
||||
### Phase 2 - GDN Follow-up
|
||||
|
||||
Do after Phase 1 unless Phase 0 proves decode/GDN is the larger immediate gap.
|
||||
|
||||
Deliverables:
|
||||
|
||||
- Clean M5 vs candidate geometry A/B.
|
||||
- Correctness gates for all candidate variants.
|
||||
|
||||
Exit criteria:
|
||||
|
||||
- Keep the best variant or close the branch with measured evidence.
|
||||
|
||||
### Phase 3 - MoE Fan-in And Glue Fusions
|
||||
|
||||
Do after kernel work identifies remaining non-kernel buckets.
|
||||
|
||||
Deliverables:
|
||||
|
||||
- nsys-backed bucket selection.
|
||||
- Fusion implementation only for material buckets.
|
||||
|
||||
Exit criteria:
|
||||
|
||||
- Keep only fusions that move end-to-end numbers beyond noise.
|
||||
|
||||
### Phase 4 - Adaptive Serving
|
||||
|
||||
Do after compute kernels are stable.
|
||||
|
||||
Deliverables:
|
||||
|
||||
- Adaptive scheduling policy.
|
||||
- Serving A/B at N=8,32,128,256.
|
||||
|
||||
Exit criteria:
|
||||
|
||||
- Improve serving without TTFT collapse.
|
||||
|
||||
## Decision Rules
|
||||
|
||||
- Prefer measured in-backend results over source plausibility.
|
||||
- Prefer small kill-gate experiments over multi-week rewrites.
|
||||
- Continue a branch only if it beats the current shipped path, not an obsolete
|
||||
baseline.
|
||||
- Document rejected branches with artifact paths so they are not rerun.
|
||||
- Keep the fork branch canonical and regenerate LocalAI patches from it.
|
||||
|
||||
@@ -1,172 +0,0 @@
|
||||
# GDN Shared-A/Ai Cost Model
|
||||
|
||||
Phase 12 decides whether the next GDN prefill attempt should implement a
|
||||
shared-A/Ai global-scratch prototype or stop GDN kernel work on GB10.
|
||||
|
||||
## Reference Points
|
||||
|
||||
llama.cpp:
|
||||
|
||||
- `/home/mudler/_git/llama.cpp/ggml/src/ggml-cuda/gated_delta_net.cu`
|
||||
- `gated_delta_net_chunked_cuda`
|
||||
- `launch_gdn_chunked`
|
||||
- `launch_gated_delta_net`
|
||||
- `ggml_cuda_op_gated_delta_net`
|
||||
|
||||
vLLM/FLA:
|
||||
|
||||
- `/home/mudler/_git/vllm/vllm/model_executor/layers/fla/ops/chunk.py`
|
||||
- `chunk_gated_delta_rule_fwd`
|
||||
- `/home/mudler/_git/vllm/vllm/model_executor/layers/fla/ops/solve_tril.py`
|
||||
- `solve_tril`
|
||||
- `solve_tril_16x16_kernel`
|
||||
- `merge_16x16_to_32x32_inverse_kernel`
|
||||
- `merge_16x16_to_64x64_inverse_kernel`
|
||||
- `/home/mudler/_git/vllm/vllm/model_executor/layers/fla/ops/wy_fast.py`
|
||||
- `recompute_w_u_fwd`
|
||||
|
||||
## Metadata
|
||||
|
||||
DGX metadata artifact:
|
||||
|
||||
- `/home/mudler/bench/phase12_gdn_shared_ai_cost_model/model_metadata.txt`
|
||||
|
||||
GGUF metadata:
|
||||
|
||||
| Model | Arch | Blocks | Full-attn interval | GDN layers | SSM inner | SSM state | GDN heads |
|
||||
|-------|------|--------|--------------------|------------|-----------|-----------|-----------|
|
||||
| MoE | `qwen35moe` | 41 | 4 | 30 inferred | 4096 | 128 | 32 inferred |
|
||||
| Dense | `qwen35` | 64 | 4 | 48 inferred | 6144 | 128 | 48 inferred |
|
||||
|
||||
Notes:
|
||||
|
||||
- `GDN heads = ssm.inner_size / ssm.state_size`.
|
||||
- MoE has one `nextn` layer; the serving/prefill stack uses the 40 normal
|
||||
layers, with 30 GDN layers at interval 4.
|
||||
- Dense has 64 layers, 48 GDN layers at interval 4.
|
||||
|
||||
## Dynamic Shared Memory
|
||||
|
||||
Formula:
|
||||
|
||||
```text
|
||||
C16 full-width current M5:
|
||||
floats = S_v*S_v + 2*C*S_v + S_v*C + C*C + 3*C + 2*C*C
|
||||
|
||||
C32 full-width:
|
||||
floats = S_v*S_v + 2*C*S_v + S_v*C + C*C + 3*C + 2*C*C
|
||||
|
||||
C32 slab64 with U staging:
|
||||
floats = S_v*64 + 2*C*S_v + 64*C + C*C + 3*C + 2*C*C + 64*C
|
||||
```
|
||||
|
||||
For `S_v=128`:
|
||||
|
||||
| Shape | Bytes | KiB | Fits GB10 dynamic smem? |
|
||||
|-------|-------|-----|-------------------------|
|
||||
| C16 full-width | 93,376 | 91.19 | yes |
|
||||
| C32 full-width | 127,360 | 124.38 | no |
|
||||
| C32 slab64 + U staging | 94,592 | 92.38 | yes |
|
||||
|
||||
Implication:
|
||||
|
||||
- C32 full-width cannot be a single current-style CTA on GB10.
|
||||
- C32 only fits by splitting value columns or by changing state residency.
|
||||
- Splitting value columns must share A/Ai or it repeats the Phase 10 failure.
|
||||
|
||||
## Ai Scratch Size
|
||||
|
||||
Formula:
|
||||
|
||||
```text
|
||||
Ai scratch bytes = npl * H * ceil(npp / BT) * BT * BT * sizeof(dtype)
|
||||
```
|
||||
|
||||
Benchmark shape: `npl=32`, `S_v=128`.
|
||||
|
||||
| Model | H | npp | BT | Ai dtype | Chunks | Ai scratch MiB | 3x Ai traffic MiB |
|
||||
|-------|---|-----|----|----------|--------|----------------|-------------------|
|
||||
| MoE | 32 | 512 | 32 | f32 | 16 | 64.0 | 192.0 |
|
||||
| MoE | 32 | 512 | 32 | f16 | 16 | 32.0 | 96.0 |
|
||||
| MoE | 32 | 512 | 64 | f32 | 8 | 128.0 | 384.0 |
|
||||
| MoE | 32 | 512 | 64 | f16 | 8 | 64.0 | 192.0 |
|
||||
| MoE | 32 | 2048 | 32 | f32 | 64 | 256.0 | 768.0 |
|
||||
| MoE | 32 | 2048 | 32 | f16 | 64 | 128.0 | 384.0 |
|
||||
| MoE | 32 | 2048 | 64 | f32 | 32 | 512.0 | 1536.0 |
|
||||
| MoE | 32 | 2048 | 64 | f16 | 32 | 256.0 | 768.0 |
|
||||
| Dense | 48 | 512 | 32 | f32 | 16 | 96.0 | 288.0 |
|
||||
| Dense | 48 | 512 | 32 | f16 | 16 | 48.0 | 144.0 |
|
||||
| Dense | 48 | 512 | 64 | f32 | 8 | 192.0 | 576.0 |
|
||||
| Dense | 48 | 512 | 64 | f16 | 8 | 96.0 | 288.0 |
|
||||
| Dense | 48 | 2048 | 32 | f32 | 64 | 384.0 | 1152.0 |
|
||||
| Dense | 48 | 2048 | 32 | f16 | 64 | 192.0 | 576.0 |
|
||||
| Dense | 48 | 2048 | 64 | f32 | 32 | 768.0 | 2304.0 |
|
||||
| Dense | 48 | 2048 | 64 | f16 | 32 | 384.0 | 1152.0 |
|
||||
|
||||
`3x Ai traffic` means one Ai write plus two Ai reads for two value slabs.
|
||||
|
||||
## Interpretation
|
||||
|
||||
The f32 `BT=32` scratch path is large but plausible:
|
||||
|
||||
- Peak scratch is 256 MiB for MoE and 384 MiB for dense at `npp=2048,npl=32`.
|
||||
- Ai traffic is 768 MiB for MoE and 1.125 GiB for dense per GDN layer call.
|
||||
- This is not free on LPDDR5x, but it is not automatically worse than
|
||||
recomputing A/Ai in every value slab.
|
||||
|
||||
The f16/BF16 Ai path halves traffic but should not be first because Phase 10 and
|
||||
Phase 11 showed correctness must be established before performance. The first
|
||||
prototype should store Ai in f32, stay default-off, and use md5/KL gates before
|
||||
trying a lossy Ai dtype.
|
||||
|
||||
## Decision
|
||||
|
||||
GO: Phase 13 should implement a default-off global-Ai scratch prototype.
|
||||
|
||||
Rationale:
|
||||
|
||||
- The only remaining C32 path that addresses Phase 10's failure is sharing A/Ai
|
||||
across value slabs.
|
||||
- `BT=32` f32 scratch has acceptable peak memory for the existing GB10
|
||||
benchmark shapes.
|
||||
- The implementation can be default-off and rejected cleanly if global scratch
|
||||
traffic or extra launch boundaries dominate.
|
||||
|
||||
Phase 13 constraints:
|
||||
|
||||
- Prototype only `BT=32`, f32 Ai, two `dv_tile=64` value slabs.
|
||||
- Keep decode out via `GDN_CHUNK_MIN > 1`.
|
||||
- Gate with `GATED_DELTA_NET`, canonical MoE/dense md5, and same-session A/B.
|
||||
- If md5 changes, run KL before benchmarking.
|
||||
- If the prototype is flat or slower, reject it and stop GDN kernel work on
|
||||
GB10; do not iterate into f16 Ai until f32 proves the schedule can win.
|
||||
|
||||
## Phase 13 Result
|
||||
|
||||
Phase 13 implemented the f32 Global-Ai32 prototype and rejected it.
|
||||
|
||||
Correctness:
|
||||
|
||||
- MoE md5: `8cb0ce23777bf55f92f63d0292c756b0`.
|
||||
- Dense md5: `5951a5b4d624ce891e22ab5fca9bc439`.
|
||||
|
||||
Performance:
|
||||
|
||||
| Model | Mode | PP | S_PP t/s |
|
||||
|-------|------|----|----------|
|
||||
| MoE | M5 base | 2048 | 2425.10 |
|
||||
| MoE | Global Ai32 | 2048 | 2097.76 |
|
||||
| Dense | M5 base | 2048 | 1016.14 |
|
||||
| Dense | Global Ai32 | 2048 | 918.19 |
|
||||
|
||||
Artifacts:
|
||||
|
||||
- `/home/mudler/bench/phase13_gdn_global_ai32/gates/`
|
||||
- `/home/mudler/bench/phase13_gdn_global_ai32/ab/`
|
||||
- `/home/mudler/bench/phase13_gdn_global_ai32/rejected/global_ai32_rejected.diff`
|
||||
|
||||
Final decision:
|
||||
|
||||
- Reject Global-Ai32.
|
||||
- Stop GDN kernel work on GB10. The remaining vLLM GDN advantage is not
|
||||
reachable through the low-conflict C16/C32 patch shapes tested here.
|
||||
@@ -1,514 +0,0 @@
|
||||
# Plan: ship the paged llama.cpp as its OWN backend + NVFP4 Qwen3.6 gallery items
|
||||
|
||||
Scoping deliverable only. NOTHING is changed by this document. It is grounded in the
|
||||
actual repo structure (read 2026-06-26 in worktree feat+paged-attention), not assumptions.
|
||||
|
||||
SHIPPED REALITY (update 2026-06-27): the backend ships CUDA-only. The matrix rows and
|
||||
the index.yaml meta-backend keep ONLY the CUDA/cublas variants (cuda-12, cuda-13, and
|
||||
the nvidia-l4t arm64 cuda-12/cuda-13 Jetson rows). The cpu / vulkan / sycl / hipblas /
|
||||
metal-darwin variants discussed below as optional/phase-2 were NOT shipped (and the
|
||||
darwin row was removed): off-CUDA the patchset's wins gate off, so it is neutral-to-
|
||||
negative there and non-CUDA users should use the stock llama-cpp backend (README 4c).
|
||||
|
||||
================================================================================
|
||||
0. GROUND TRUTH (what the repo actually does today)
|
||||
================================================================================
|
||||
|
||||
The paged patchset is ALREADY integrated into the stock llama-cpp backend in this
|
||||
worktree. Two mechanisms, both already present:
|
||||
|
||||
(a) BUILD: backend/cpp/llama-cpp/Makefile has `LLAMA_PAGED?=on`. The `llama.cpp:`
|
||||
target git-applies patches/0*.patch (base series) then, when LLAMA_PAGED != off,
|
||||
patches/paged/0*.patch (the 0018-0023 paged series + the earlier 0001-0017).
|
||||
prepare.sh has a fallback `patch`-based apply guarded by a sentinel
|
||||
(llama.cpp/src/paged-kv-manager.cpp). So a stock `make backends/llama-cpp` TODAY
|
||||
already ships the paged engine compiled in.
|
||||
|
||||
(b) RUNTIME GATING: backend/cpp/llama-cpp/grpc-server.cpp ALREADY carries the option
|
||||
hooks (lines ~752-842). They only call setenv() before context init:
|
||||
- option `kv_paged` / `paged_kv` / `paged_attention` -> setenv LLAMA_KV_PAGED=1
|
||||
- option `kv_paged_debug` / `paged_kv_debug` -> setenv LLAMA_KV_PAGED_DEBUG=1
|
||||
- option `max_prefill_tokens` / `mpt` / `prefill_budget` -> setenv LLAMA_PREFILL_BUDGET
|
||||
- option `max_batch_tokens` / `mbt` -> setenv LLAMA_MAX_BATCH_TOKENS
|
||||
- option `prefill_cap` -> setenv LLAMA_PREFILL_CAP
|
||||
Against UNPATCHED llama.cpp these setenv() calls are inert (nothing reads the env),
|
||||
so grpc-server.cpp is byte-safe to share between a clean build and a paged build.
|
||||
The paged engine itself lives entirely inside the patched llama.cpp lib
|
||||
(paged-kv-manager.cpp etc.), NOT in grpc-server.cpp.
|
||||
|
||||
Conclusion: "stock llama-cpp + paged patchset, runtime-gated" is the CURRENT state of
|
||||
ONE backend. The task is to SPLIT that into two backends:
|
||||
- llama-cpp = clean upstream llama.cpp (de-risked: a dep-bump can never break on a
|
||||
paged hook), grpc-server.cpp keeps the dormant hooks.
|
||||
- <newname> = stock grpc-server.cpp + paged patch series applied + paged on.
|
||||
|
||||
The turboquant backend is the EXACT precedent for "a llama.cpp variant that reuses the
|
||||
backend/cpp/llama-cpp grpc-server sources via a thin wrapper Makefile + its own Dockerfile
|
||||
+ its own matrix rows". Copy turboquant's shape, with two simplifications (see section 1).
|
||||
|
||||
CPU_ALL_VARIANTS reuse: backend/cpp/llama-cpp/Makefile already has `llama-cpp-cpu-all`
|
||||
(one grpc-server + dlopen libggml-cpu-*.so via -DGGML_BACKEND_DL/-DGGML_CPU_ALL_VARIANTS,
|
||||
SHARED_LIBS=ON make-var). turboquant mirrors it with `turboquant-cpu-all`. The new backend
|
||||
gets the same single-build CPU target for free by reusing the same Makefile machinery.
|
||||
|
||||
--------------------------------------------------------------------------------
|
||||
RECOMMENDED BACKEND NAME: `llama-cpp-paged` (see section 4 for the full rationale)
|
||||
--------------------------------------------------------------------------------
|
||||
Everywhere below, NAME = llama-cpp-paged, DOCKERFILE = Dockerfile.llama-cpp-paged,
|
||||
SRC DIR = backend/cpp/llama-cpp-paged/, MAKE VAR = BACKEND_LLAMA_CPP_PAGED.
|
||||
DO NOT use the dotted working name `localai-llama.cpp`: a dot in Dockerfile.<suffix> and
|
||||
in the tag-suffix is unprecedented (every sibling is hyphenated: llama-cpp, ik-llama-cpp,
|
||||
turboquant, ds4) and complicates the changed-backends.js endsWith() suffix matching.
|
||||
|
||||
================================================================================
|
||||
1. NEW BACKEND - file by file
|
||||
================================================================================
|
||||
|
||||
--------------------------------------------------------------------------------
|
||||
1.1 backend/cpp/llama-cpp/Makefile (the ONE necessary touch to stock)
|
||||
--------------------------------------------------------------------------------
|
||||
Change exactly one default so the STOCK image ships clean against upstream:
|
||||
|
||||
-LLAMA_PAGED?=on
|
||||
+LLAMA_PAGED?=off
|
||||
|
||||
Why: this is the entire point of the split - stock llama-cpp must build clean so an
|
||||
upstream LLAMA_VERSION bump can never fail on a paged hook. The runtime hooks in
|
||||
grpc-server.cpp stay (inert). The new backend forces LLAMA_PAGED=on explicitly (1.2), so
|
||||
it does not depend on this default. NOTE this DOES change stock's shipped artifact (it
|
||||
currently ships paged-compiled-in-but-gated); that is intended de-risking, call it out in
|
||||
the PR. If the team prefers stock literally untouched, the alternative is to leave
|
||||
`?=on` and accept that stock keeps carrying the patch series - but then "clean stock" is
|
||||
not achieved. Recommendation: flip to off.
|
||||
|
||||
(No other change to backend/cpp/llama-cpp/ - grpc-server.cpp, CMakeLists.txt, prepare.sh,
|
||||
patches/, patches/paged/ are all reused as-is by the new backend.)
|
||||
|
||||
--------------------------------------------------------------------------------
|
||||
1.2 backend/cpp/llama-cpp-paged/Makefile (NEW - thin wrapper, model on turboquant)
|
||||
--------------------------------------------------------------------------------
|
||||
Mirror backend/cpp/turboquant/Makefile, but SIMPLER (two things turboquant needs that we
|
||||
do NOT):
|
||||
- turboquant overrides LLAMA_REPO/LLAMA_VERSION to a fork. We use the SAME upstream pin
|
||||
as stock (it lives in backend/cpp/llama-cpp/Makefile, already auto-bumped). So we do
|
||||
NOT set LLAMA_VERSION here -> no bump_deps.yaml entry needed (big simplification vs
|
||||
turboquant). We only force LLAMA_PAGED=on.
|
||||
- turboquant runs patch-grpc-server.sh (augments the KV-cache type allow-list) and
|
||||
apply-patches.sh (fork catch-up). We need NEITHER: grpc-server.cpp already has the
|
||||
paged hooks, and the paged patch series is applied by the copied llama-cpp Makefile's
|
||||
own `llama.cpp:` target when LLAMA_PAGED=on.
|
||||
|
||||
Shape (one flavor shown; replicate the turboquant flavor set: avx/avx2/avx512/fallback/
|
||||
cpu-all/grpc/rpc-server):
|
||||
|
||||
LLAMA_CPP_DIR := $(CURRENT_MAKEFILE_DIR)/../llama-cpp
|
||||
|
||||
define paged-build # $(1)=flavor $(2)=cmake flags $(3)=target
|
||||
rm -rf $(CURRENT_MAKEFILE_DIR)/../llama-cpp-paged-$(1)-build
|
||||
cp -rf $(LLAMA_CPP_DIR) $(CURRENT_MAKEFILE_DIR)/../llama-cpp-paged-$(1)-build
|
||||
$(MAKE) -C $(CURRENT_MAKEFILE_DIR)/../llama-cpp-paged-$(1)-build purge
|
||||
# clone upstream + apply base AND paged patch series (LLAMA_PAGED=on forces it)
|
||||
LLAMA_PAGED=on $(MAKE) -C $(CURRENT_MAKEFILE_DIR)/../llama-cpp-paged-$(1)-build llama.cpp
|
||||
CMAKE_ARGS="$(CMAKE_ARGS) $(2)" TARGET="$(3)" LLAMA_PAGED=on \
|
||||
$(MAKE) -C $(CURRENT_MAKEFILE_DIR)/../llama-cpp-paged-$(1)-build grpc-server
|
||||
cp -rfv $(CURRENT_MAKEFILE_DIR)/../llama-cpp-paged-$(1)-build/grpc-server llama-cpp-paged-$(1)
|
||||
endef
|
||||
|
||||
llama-cpp-paged-cpu-all:
|
||||
# identical to turboquant-cpu-all: SHARED_LIBS=ON + GGML_BACKEND_DL + CPU_ALL_VARIANTS
|
||||
# + --target ggml; then collect ggml-shared-libs/ for package.sh to bundle.
|
||||
... LLAMA_PAGED=on SHARED_LIBS=ON \
|
||||
EXTRA_CMAKE_ARGS="-DGGML_BACKEND_DL=ON -DGGML_CPU_ALL_VARIANTS=ON" \
|
||||
TARGET="--target grpc-server --target ggml" ...
|
||||
|
||||
package: ; bash package.sh
|
||||
purge: ; rm -rf $(CURRENT_MAKEFILE_DIR)/../llama-cpp-paged-*-build; rm -rf llama-cpp-paged-* package
|
||||
clean: purge
|
||||
|
||||
Binaries are named llama-cpp-paged-{cpu-all,fallback,grpc,rpc-server,...} so run.sh and
|
||||
package.sh glob them.
|
||||
|
||||
--------------------------------------------------------------------------------
|
||||
1.3 backend/cpp/llama-cpp-paged/run.sh (NEW - copy turboquant/run.sh, rename binaries)
|
||||
--------------------------------------------------------------------------------
|
||||
s/turboquant/llama-cpp-paged/g. Prefers llama-cpp-paged-cpu-all if present, falls back to
|
||||
llama-cpp-paged-fallback; llama-cpp-paged-grpc when LLAMACPP_GRPC_SERVERS set; Darwin
|
||||
DYLD_LIBRARY_PATH branch; lib/ld.so launch. Keep verbatim otherwise.
|
||||
|
||||
--------------------------------------------------------------------------------
|
||||
1.4 backend/cpp/llama-cpp-paged/package.sh (NEW - copy turboquant/package.sh, rename)
|
||||
--------------------------------------------------------------------------------
|
||||
s/turboquant/llama-cpp-paged/g. Copies llama-cpp-paged-* into package/, bundles
|
||||
ggml-shared-libs/*.so* into package/lib (the CPU_ALL_VARIANTS dlopen set), copies run.sh,
|
||||
and the per-arch libc/ld.so set (unchanged).
|
||||
|
||||
--------------------------------------------------------------------------------
|
||||
1.5 backend/Dockerfile.llama-cpp-paged (NEW - copy Dockerfile.turboquant, swap paths)
|
||||
--------------------------------------------------------------------------------
|
||||
Identical 3-stage structure (builder-fromsource / builder-prebuilt / FROM scratch). Edits:
|
||||
- bind/run .docker/llama-cpp-paged-compile.sh (new, 1.6) instead of turboquant-compile.sh
|
||||
- ccache id: id=llama-cpp-paged-ccache-${TARGETARCH}-${BUILD_TYPE}
|
||||
(OPTIONAL OPTIMIZATION: set id=llama-cpp-ccache-${TARGETARCH}-${BUILD_TYPE} to SHARE
|
||||
stock llama-cpp's ccache - the paged TUs are mostly byte-identical to stock, so a warm
|
||||
stock cache would give the paged build near-free object reuse. Trade-off: a regression
|
||||
in one could surface as a cold miss in the other. Recommend sharing; revisit if noisy.)
|
||||
- both `make -BC /LocalAI/backend/cpp/llama-cpp-paged package`
|
||||
- final COPY --from=builder /LocalAI/backend/cpp/llama-cpp-paged/package/. ./
|
||||
|
||||
--------------------------------------------------------------------------------
|
||||
1.6 .docker/llama-cpp-paged-compile.sh (NEW - copy llama-cpp-compile.sh, swap make targets)
|
||||
--------------------------------------------------------------------------------
|
||||
Identical to .docker/llama-cpp-compile.sh except `cd .../llama-cpp-paged` and call
|
||||
`make llama-cpp-paged-cpu-all` (BUILD_TYPE empty / CPU) or `make llama-cpp-paged-fallback`
|
||||
(GPU), then `make llama-cpp-paged-grpc` + `make llama-cpp-paged-rpc-server`. Keep the
|
||||
arm64 gcc-14 apt step (CPU_ALL_VARIANTS armv9.2 SME needs gcc-14). ccache export unchanged.
|
||||
|
||||
--------------------------------------------------------------------------------
|
||||
1.7 Makefile (top-level) - 6 edits, mirror the turboquant lines
|
||||
--------------------------------------------------------------------------------
|
||||
a) .NOTPARALLEL (line 2): append `backends/llama-cpp-paged`
|
||||
b) Backend def (after BACKEND_TURBOQUANT, line ~1172):
|
||||
# llama-cpp-paged = stock llama.cpp grpc-server + LocalAI paged-attention patch
|
||||
# series (LLAMA_PAGED=on). Reuses backend/cpp/llama-cpp sources via a thin wrapper.
|
||||
BACKEND_LLAMA_CPP_PAGED = llama-cpp-paged|llama-cpp-paged|.|false|false
|
||||
(lang field `llama-cpp-paged` -> Dockerfile.llama-cpp-paged, matching the
|
||||
llama-cpp / ik-llama-cpp / turboquant convention where lang==backend name.)
|
||||
c) generate-docker-build-target eval (after BACKEND_TURBOQUANT, line ~1273):
|
||||
$(eval $(call generate-docker-build-target,$(BACKEND_LLAMA_CPP_PAGED)))
|
||||
d) docker-build-backends (line ~1337): append docker-build-llama-cpp-paged
|
||||
e) test-extra-backend-llama-cpp-paged target (mirror test-extra-backend-turboquant,
|
||||
line ~673): BACKEND_IMAGE=local-ai-backend:llama-cpp-paged $(MAKE) test-extra-backend
|
||||
f) (optional) backends/llama-cpp-paged-darwin target if shipping metal (mirror
|
||||
backends/llama-cpp-darwin at line 1124; see 1.11).
|
||||
|
||||
--------------------------------------------------------------------------------
|
||||
1.8 .github/backend-matrix.yml - add rows (mirror every llama-cpp row, swap names)
|
||||
--------------------------------------------------------------------------------
|
||||
For EACH variant you choose to ship (see phased recommendation in section 4), add a row
|
||||
copied from the corresponding llama-cpp row with:
|
||||
- backend: "llama-cpp-paged"
|
||||
- dockerfile: "./backend/Dockerfile.llama-cpp-paged"
|
||||
- tag-suffix: swap `-llama-cpp` -> `-llama-cpp-paged`
|
||||
(e.g. -cpu-llama-cpp -> -cpu-llama-cpp-paged;
|
||||
-gpu-nvidia-cuda-12-llama-cpp -> -gpu-nvidia-cuda-12-llama-cpp-paged; etc.)
|
||||
- builder-base-image: UNCHANGED - reuse the same base-grpc-* tags as llama-cpp
|
||||
(this backend compiles the same gRPC + same toolchain; no new base-images.yml variant
|
||||
is needed, so NO base-images bootstrap step). This is the cheap-variant payoff.
|
||||
- CPU: TWO per-arch rows (amd64 ubuntu-latest + arm64 ubuntu-24.04-arm) sharing
|
||||
tag-suffix '-cpu-llama-cpp-paged' so changed-backends.js emits a merge-matrix entry and
|
||||
backend-merge-jobs assembles the manifest list. Same per-arch native + manifest-merge
|
||||
pattern as -cpu-llama-cpp.
|
||||
- Darwin (if shipping): add to includeDarwin:
|
||||
- backend: "llama-cpp-paged"
|
||||
tag-suffix: "-metal-darwin-arm64-llama-cpp-paged"
|
||||
lang: "go"
|
||||
(omit build-type, exactly like the llama-cpp darwin row at line 4908.)
|
||||
|
||||
REMINDER: the CI path filter only builds a backend on a PR when a file under its dir
|
||||
changes. The PR that adds this backend touches backend/cpp/llama-cpp-paged/* so it self-
|
||||
triggers. But also add the cross-trigger in 1.9 so future edits to backend/cpp/llama-cpp/
|
||||
(the shared source) retrigger this backend too.
|
||||
|
||||
--------------------------------------------------------------------------------
|
||||
1.9 scripts/changed-backends.js - two edits (mirror turboquant exactly)
|
||||
--------------------------------------------------------------------------------
|
||||
a) inferBackendPath(): add BEFORE the generic `endsWith("llama-cpp")` branch (line 56),
|
||||
next to the turboquant branch (line 45):
|
||||
if (item.dockerfile.endsWith("llama-cpp-paged")) {
|
||||
// reuses backend/cpp/llama-cpp sources via a thin wrapper Makefile
|
||||
return `backend/cpp/llama-cpp-paged/`;
|
||||
}
|
||||
ORDER MATTERS: "Dockerfile.llama-cpp-paged".endsWith("llama-cpp") is false today, but
|
||||
keep the specific branch first regardless (defensive, and returns the right path).
|
||||
b) inferBackendPathDarwin(): add a case (next to the llama-cpp one at line 66):
|
||||
if (item.backend === "llama-cpp-paged") { return `backend/cpp/llama-cpp-paged/`; }
|
||||
c) Per-backend cross-trigger (line 274-278, mirror the turboquant block):
|
||||
if (backend === "llama-cpp-paged" && !changed) {
|
||||
changed = changedFiles.some(file => file.startsWith("backend/cpp/llama-cpp/"));
|
||||
}
|
||||
Verify: node -e "... e.dockerfile.endsWith('llama-cpp-paged') ..." per adding-backends.md.
|
||||
|
||||
--------------------------------------------------------------------------------
|
||||
1.10 backend/index.yaml - meta + image entries (META-BACKEND - capabilities map, NO uri)
|
||||
--------------------------------------------------------------------------------
|
||||
GOTCHA (project_backend_meta_gotcha): a backend that ships per-platform images MUST be a
|
||||
meta backend = an anchor with a `capabilities:` map and NO top-level `uri:`; the concrete
|
||||
per-platform entries carry the uri. Copy the *llamacpp anchor (lines 3-31).
|
||||
|
||||
Step a - meta anchor in `## metas` (after *turboquant, ~line 74):
|
||||
- &llamacpppaged
|
||||
name: "llama-cpp-paged"
|
||||
alias: "llama-cpp-paged"
|
||||
license: mit
|
||||
icon: <same as llama-cpp>
|
||||
description: |
|
||||
LocalAI's paged-attention llama.cpp: on-demand paged KV cache + decode-first
|
||||
prefill budget. Stock llama.cpp grpc-server + the LocalAI paged patch series.
|
||||
Tuned for NVFP4 dense/MoE on Blackwell/GB10. Reuses the llama-cpp gRPC server.
|
||||
urls: [ https://github.com/ggerganov/llama.cpp ]
|
||||
tags: [ text-to-text, LLM, CPU, GPU, CUDA, Metal, paged-attention, nvfp4 ]
|
||||
capabilities:
|
||||
default: "cpu-llama-cpp-paged"
|
||||
nvidia: "cuda12-llama-cpp-paged"
|
||||
nvidia-cuda-12: "cuda12-llama-cpp-paged"
|
||||
nvidia-cuda-13: "cuda13-llama-cpp-paged"
|
||||
nvidia-l4t: "nvidia-l4t-arm64-llama-cpp-paged"
|
||||
nvidia-l4t-cuda-12: "nvidia-l4t-arm64-llama-cpp-paged"
|
||||
nvidia-l4t-cuda-13: "cuda13-nvidia-l4t-arm64-llama-cpp-paged"
|
||||
metal: "metal-llama-cpp-paged"
|
||||
# add amd/intel/vulkan keys ONLY for variants you actually build (section 4)
|
||||
|
||||
Step b - a `-development` meta (mirror llama-cpp-development, line 1611) with the same
|
||||
capabilities map pointing at the `*-development` image names.
|
||||
|
||||
Step c - concrete image entries at end of file (mirror the llama-cpp block lines
|
||||
2106-2200), one latest + one development per variant, each as:
|
||||
- !!merge <<: *llamacpppaged
|
||||
name: "cpu-llama-cpp-paged"
|
||||
uri: "quay.io/go-skynet/local-ai-backends:latest-cpu-llama-cpp-paged"
|
||||
mirrors: [ localai/localai-backends:latest-cpu-llama-cpp-paged ]
|
||||
- !!merge <<: *llamacpppaged
|
||||
name: "cpu-llama-cpp-paged-development"
|
||||
uri: "quay.io/go-skynet/local-ai-backends:master-cpu-llama-cpp-paged"
|
||||
mirrors: [ localai/localai-backends:master-cpu-llama-cpp-paged ]
|
||||
...repeat for cuda12 / cuda13 / l4t / metal etc.
|
||||
The `latest-` / `master-` uri prefix + tag-suffix MUST match the matrix tag-suffix exactly.
|
||||
|
||||
--------------------------------------------------------------------------------
|
||||
1.11 Darwin (only if shipping metal; the NVFP4 target is CUDA, so metal is optional/phase 2)
|
||||
--------------------------------------------------------------------------------
|
||||
If metal is shipped, also:
|
||||
- scripts/build/llama-cpp-paged-darwin.sh (copy scripts/build/llama-cpp-darwin.sh; it
|
||||
drives the 3 CMake variants + otool dylib bundling). Ensure it forces LLAMA_PAGED=on.
|
||||
- Makefile `backends/llama-cpp-paged-darwin` target (mirror backends/llama-cpp-darwin).
|
||||
- backend_build_darwin.yml: add the llama-cpp-paged branch (mirror the llama-cpp-specific
|
||||
step that calls `make backends/llama-cpp-darwin`).
|
||||
- index.yaml metal-llama-cpp-paged / -development image entries (already in 1.10).
|
||||
- C++ proto gotcha already handled (reuses llama-cpp CMakeLists.txt with hw_grpc_proto
|
||||
linking protobuf/grpc++), so no Homebrew-include failure.
|
||||
|
||||
--------------------------------------------------------------------------------
|
||||
1.12 Importer / /backends/known dropdown (drop-in, NOT a new importer)
|
||||
--------------------------------------------------------------------------------
|
||||
This backend consumes GGUF exactly like llama-cpp -> extend the EXISTING importer, do not
|
||||
add a new one (per adding-backends.md rule 2). Edit core/gallery/importers/llama-cpp.go:
|
||||
- AdditionalBackends() (line 37): append
|
||||
{Name: "llama-cpp-paged", Modality: "text",
|
||||
Description: "Paged-attention llama.cpp (on-demand paged KV + decode-first budget)"}
|
||||
- Import() backend allow-list (line 133): add "llama-cpp-paged" to the switch case so a
|
||||
preferences.backend == "llama-cpp-paged" is honored:
|
||||
case "ik-llama-cpp", "turboquant", "llama-cpp-paged": backend = b
|
||||
- core/gallery/importers/importers_test.go: add a table case asserting the preference
|
||||
override emits backend: llama-cpp-paged (Ginkgo/Gomega; reuse an existing public GGUF
|
||||
HF fixture). Run `go test ./core/gallery/importers/...`.
|
||||
|
||||
--------------------------------------------------------------------------------
|
||||
1.13 Docs
|
||||
--------------------------------------------------------------------------------
|
||||
- docs/content/features/backends.md: add llama-cpp-paged to the text-to-text/LLM list,
|
||||
one line noting paged KV + NVFP4 Blackwell tuning. (Not an in-house from-scratch engine
|
||||
-> it is a llama.cpp variant -> do NOT add to the README maintained-engines table.)
|
||||
|
||||
--------------------------------------------------------------------------------
|
||||
1.14 Does grpc-server.cpp need the paged hooks? YES - already present, reused unchanged.
|
||||
--------------------------------------------------------------------------------
|
||||
The hooks (kv_paged / max_batch_tokens / prefill_budget / prefill_cap) are already in the
|
||||
SHARED backend/cpp/llama-cpp/grpc-server.cpp. The paged backend reuses that file verbatim
|
||||
(via the Makefile copy). No patch-grpc-server.sh step is needed (unlike turboquant). The
|
||||
hooks are what translate the gallery `options:` (1.10 section 2) into the LLAMA_KV_PAGED /
|
||||
LLAMA_MAX_BATCH_TOKENS env that the paged llama.cpp lib reads.
|
||||
|
||||
================================================================================
|
||||
2. GALLERY ITEMS - NVFP4 Qwen3.6 dense + MoE
|
||||
================================================================================
|
||||
|
||||
Add two entries to gallery/index.yaml. Schema (verified against existing GGUF items and
|
||||
the LocalAI config structs): backend selection via `overrides.backend`; runtime knobs via
|
||||
either typed config fields (context_size/f16/flash_attention/gpu_layers/batch) or the
|
||||
`options:` string list (key:value, parsed by grpc-server.cpp set_option).
|
||||
|
||||
--------------------------------------------------------------------------------
|
||||
2.1 Benchmark llama-server flags -> LocalAI model-config mapping
|
||||
--------------------------------------------------------------------------------
|
||||
-c 131072 -> context_size: 131072 (LLMConfig.ContextSize, yaml context_size)
|
||||
-fa on -> flash_attention: "on" (LLMConfig.FlashAttention, yaml flash_attention; string)
|
||||
-ngl 99 -> gpu_layers: 99 (LLMConfig.NGPULayers, yaml gpu_layers; or omit -> DefaultNGPULayers offloads all)
|
||||
-b 2048 -> batch: 2048 (schema.PredictionOptions.Batch, yaml batch) [see caveat]
|
||||
--parallel 128 -> options: ["parallel:128"] (grpc-server.cpp:629; alias n_parallel)
|
||||
LLAMA_KV_PAGED=1 -> options: ["paged_kv:true"] (grpc-server.cpp:778)
|
||||
LLAMA_MAX_BATCH_TOKENS=512 -> options: ["max_batch_tokens:512"] (grpc-server.cpp:821; alias mbt)
|
||||
f16 KV -> f16: true (LLMConfig.F16, yaml f16)
|
||||
(recommended for paged) -> options: ["kv_unified:false"] (grpc-server.cpp:746 - the per-slot paged
|
||||
capacity/memory benefit only materializes with a per-sequence cache;
|
||||
the patch comment explicitly recommends pairing paged with kv_unified:false)
|
||||
|
||||
CAVEAT (-ub 512): LocalAI sets params.n_ubatch = params.n_batch = request->nbatch()
|
||||
(grpc-server.cpp:528,532). There is NO separate config field for n_ubatch, so the
|
||||
benchmark's `-b 2048 -ub 512` split is NOT exactly reproducible. Options:
|
||||
(i) set batch: 512 -> n_batch=n_ubatch=512 (matches -ub; the decode-first
|
||||
max_batch_tokens=512 budget is the dominant prefill lever anyway, and the
|
||||
benchmark states decode throughput is budget-independent), OR
|
||||
(ii) set batch: 2048 -> n_ubatch also 2048 (bigger physical batch, more KV scratch).
|
||||
RECOMMEND (i) batch: 512 for the shipped gallery config (closest to the measured run +
|
||||
lighter memory). Flag separately: a tiny grpc-server.cpp option `n_ubatch`/`ubatch` could
|
||||
be added later to honor -b/-ub independently (not required to ship).
|
||||
|
||||
--------------------------------------------------------------------------------
|
||||
2.2 gallery/index.yaml entry - DENSE q36-27b-nvfp4
|
||||
--------------------------------------------------------------------------------
|
||||
- name: "qwen3.6-27b-nvfp4-paged"
|
||||
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
|
||||
urls:
|
||||
- https://huggingface.co/<ORG>/Qwen3.6-27B-NVFP4-GGUF # placeholder, section 3
|
||||
description: |
|
||||
Qwen3.6-27B dense, native Blackwell NVFP4 (FP4-MMA) GGUF. Configured for LocalAI's
|
||||
paged-attention llama.cpp backend: on-demand paged KV + decode-first prefill budget.
|
||||
Benchmarked on GB10/DGX Spark at 90-117% of vLLM dense decode at 1.5-3x lower memory.
|
||||
license: "apache-2.0" # confirm vs Qwen license
|
||||
tags: [ llm, gguf, nvfp4, reasoning ]
|
||||
icon: https://user-images.githubusercontent.com/1991296/230134379-7181e485-c521-4d23-a0d6-f7b3b61ba524.png
|
||||
overrides:
|
||||
backend: llama-cpp-paged
|
||||
f16: true
|
||||
flash_attention: "on"
|
||||
context_size: 131072
|
||||
gpu_layers: 99
|
||||
batch: 512 # see -ub caveat 2.1; matches the 512 ubatch floor
|
||||
known_usecases: [ chat ]
|
||||
options:
|
||||
- use_jinja:true
|
||||
- paged_kv:true # LLAMA_KV_PAGED=1
|
||||
- max_batch_tokens:512 # LLAMA_MAX_BATCH_TOKENS=512 (decode-first QoS budget)
|
||||
- kv_unified:false # enables the per-slot paged capacity/memory benefit
|
||||
- parallel:128 # --parallel 128 serving slots
|
||||
parameters:
|
||||
model: llama-cpp/models/Qwen3.6-27B-NVFP4-GGUF/q36-27b-nvfp4.gguf
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
files:
|
||||
- filename: llama-cpp/models/Qwen3.6-27B-NVFP4-GGUF/q36-27b-nvfp4.gguf
|
||||
sha256: <FILL after publish>
|
||||
uri: https://huggingface.co/<ORG>/Qwen3.6-27B-NVFP4-GGUF/resolve/main/q36-27b-nvfp4.gguf
|
||||
|
||||
--------------------------------------------------------------------------------
|
||||
2.3 gallery/index.yaml entry - MoE q36-35b-a3b-nvfp4
|
||||
--------------------------------------------------------------------------------
|
||||
Same shape; the MoE is lighter on memory (~3B active). parallel:128 + budget 256 was the
|
||||
MoE decode-throughput sweet spot in the sweep, but 512 is fine as a default; if optimizing
|
||||
purely for saturated MoE decode use max_batch_tokens:256.
|
||||
- name: "qwen3.6-35b-a3b-nvfp4-paged"
|
||||
urls: [ https://huggingface.co/<ORG>/Qwen3.6-35B-A3B-NVFP4-GGUF ]
|
||||
...
|
||||
overrides:
|
||||
backend: llama-cpp-paged
|
||||
f16: true
|
||||
flash_attention: "on"
|
||||
context_size: 131072
|
||||
batch: 512
|
||||
options:
|
||||
- use_jinja:true
|
||||
- paged_kv:true
|
||||
- max_batch_tokens:512 # or 256 for max saturated MoE decode (sweep winner)
|
||||
- kv_unified:false
|
||||
- parallel:128
|
||||
parameters:
|
||||
model: llama-cpp/models/Qwen3.6-35B-A3B-NVFP4-GGUF/q36-35b-a3b-nvfp4.gguf
|
||||
files:
|
||||
- filename: llama-cpp/models/Qwen3.6-35B-A3B-NVFP4-GGUF/q36-35b-a3b-nvfp4.gguf
|
||||
sha256: <FILL after publish>
|
||||
uri: https://huggingface.co/<ORG>/Qwen3.6-35B-A3B-NVFP4-GGUF/resolve/main/q36-35b-a3b-nvfp4.gguf
|
||||
|
||||
Note: these are the BENCHMARK serving configs. For an interactive single-user default you
|
||||
may want a second lighter gallery variant (context_size 16384, parallel 4, drop the budget)
|
||||
- optional, not required to ship the benchmark reproduction.
|
||||
|
||||
================================================================================
|
||||
3. GGUF PUBLISHING (so the gallery uri: resolves)
|
||||
================================================================================
|
||||
|
||||
The two GGUFs already exist on the DGX dev box (final_benchmark.csv references
|
||||
q36-27b-nvfp4.gguf and q36-35b-a3b-nvfp4.gguf; README.md "Models" + "Benchmarks"
|
||||
document provenance: dense = native Blackwell FP4 unsloth W4A4 lineage; MoE = 241 NVFP4
|
||||
tensors from nvidia modelopt weights). To publish:
|
||||
|
||||
1. HF repos (suggest two, under the org that owns the gallery-referenced weights):
|
||||
<ORG>/Qwen3.6-27B-NVFP4-GGUF (single q36-27b-nvfp4.gguf)
|
||||
<ORG>/Qwen3.6-35B-A3B-NVFP4-GGUF (single q36-35b-a3b-nvfp4.gguf)
|
||||
ORG = localai-org (brand) or mudler (personal); pick per ownership of the conversions.
|
||||
2. Upload each .gguf; compute sha256 (sha256sum) and paste into the gallery `files:` sha256
|
||||
(LocalAI verifies it on download). Without sha256 the entry still works but loses the
|
||||
integrity check - fill it.
|
||||
3. Model card metadata: base_model Qwen/Qwen3.6-*, library_name gguf, quantization NVFP4,
|
||||
pipeline_tag text-generation, license (confirm Qwen3.6 license terms - apache-2.0 vs
|
||||
Qwen community license), a note that it REQUIRES the llama-cpp-paged backend (NVFP4 +
|
||||
paged), and the GB10 benchmark table (link README.md "Benchmarks" numbers).
|
||||
4. NVFP4 requires a llama.cpp new enough to read the NVFP4 GGUF type. Confirm the pinned
|
||||
LLAMA_VERSION in backend/cpp/llama-cpp/Makefile supports NVFP4 tensor types (the dev
|
||||
tree that produced the GGUFs did). If the current pin predates NVFP4 GGUF support, the
|
||||
backend pin must be bumped OR the paged patch series must carry the NVFP4 reader. THIS
|
||||
IS A GATING CHECK before the gallery items are usable - verify on a GPU box.
|
||||
5. Provenance/licensing: the dense conversion derives from unsloth; the MoE from nvidia
|
||||
modelopt weights. Ensure redistribution of the converted GGUFs is permitted and
|
||||
attribute upstream in the card.
|
||||
|
||||
================================================================================
|
||||
4. OPEN DECISIONS / BLOCKERS / BUILD COST
|
||||
================================================================================
|
||||
|
||||
BACKEND NAME - RECOMMEND `llama-cpp-paged`.
|
||||
- llama-cpp-paged (RECOMMENDED): descriptive (it IS the paged variant), hyphenated like
|
||||
every sibling (llama-cpp/ik-llama-cpp/turboquant/ds4), collision-free in the
|
||||
changed-backends.js endsWith() suffix scheme, self-documenting in the /backends/known
|
||||
importer dropdown. Reads correctly next to "turboquant" and "ik-llama-cpp".
|
||||
- localai-llama-cpp (branding alternative, ACCEPTABLE): keeps the LocalAI brand without a
|
||||
dot; hyphenated and safe. Use this if marketing wants "LocalAI's own llama.cpp" framing.
|
||||
Slightly less self-explanatory about WHAT differs (paged) in the dropdown.
|
||||
- localai-llama.cpp (the working name; NOT RECOMMENDED): the dot makes Dockerfile.localai-
|
||||
llama.cpp and tag-suffix -cpu-localai-llama.cpp the only dotted ones in the repo, and
|
||||
".cpp" looks like a file extension to the suffix matcher. Avoid.
|
||||
|
||||
BLOCKERS / GATING CHECKS (cannot be closed read-only, no GPU here):
|
||||
1. NVFP4 GGUF read support in the pinned LLAMA_VERSION (section 3.4). Must verify on GPU.
|
||||
If unsupported, bump the pin (which also affects stock llama-cpp) or carry the reader.
|
||||
2. The two GGUFs are not yet on HF (section 3). Gallery uri + sha256 are placeholders
|
||||
until upload. Blocks gallery validation only, not the backend build.
|
||||
3. -ub vs -b split (section 2.1) is not exactly reproducible without a tiny grpc-server
|
||||
option; shipped config uses batch:512. Minor, not a blocker.
|
||||
4. Flipping stock LLAMA_PAGED?=off changes stock's shipped artifact (de-risking, intended)
|
||||
- get explicit sign-off since it alters a heavily-used backend's build.
|
||||
|
||||
PLATFORM SHIP MATRIX (RECOMMENDED PHASING - the variant is cheap because it reuses the same
|
||||
base-grpc-* prebuilt bases and the same compile machinery, so each row is just CI minutes):
|
||||
Phase 1 (the benchmark target - GB10/Blackwell is CUDA):
|
||||
- cuda12 amd64, cuda13 amd64, cuda13 arm64 (sbsa), l4t-cuda-12 arm64 (NVFP4/paged win)
|
||||
- cpu-all amd64 + cpu-all arm64 (the single CPU_ALL_VARIANTS build; baseline coverage)
|
||||
Phase 2 (parity with stock llama-cpp coverage, only if demand):
|
||||
- metal-darwin-arm64 (1.11), vulkan amd64/arm64, rocm amd64, intel sycl f16/f32
|
||||
Defer rocm/sycl/vulkan/metal unless asked - the paged + NVFP4 story is GPU/CUDA-centric
|
||||
and these add CI cost without a clear consumer.
|
||||
|
||||
BUILD-COST ESTIMATE PER PLATFORM (with warm base-grpc-* base + ccache; the paged TUs are
|
||||
~byte-identical to stock so a SHARED ccache id makes most objects free):
|
||||
- CPU_ALL_VARIANTS (per arch): ~15-30 min warm / ~35-50 min cold. arm64 adds a gcc-14
|
||||
apt step. Two arches + a merge job.
|
||||
- CUDA (per arch): ~25-45 min warm / ~45-75 min cold (nvcc dominates; ccache helps less
|
||||
across CUDA arch flag changes). amd64 cuda12 + cuda13, arm64 cuda13 + l4t = 4 jobs.
|
||||
- Metal/Darwin (if Phase 2): native macos-14 runner, ~20-35 min with the ccache cache.
|
||||
- No base-images.yml change and no bootstrap dispatch (reuses existing base-grpc-* tags),
|
||||
so the only new CI cost is the per-row build minutes above. PR builds read cache, don't
|
||||
write; first master build per row pays the cold cost once, then warm.
|
||||
|
||||
VERIFICATION (post-implementation, needs a GPU box - out of scope here):
|
||||
- `make backends/llama-cpp-paged` builds + installs locally (from-source path).
|
||||
- Confirm stock `make backends/llama-cpp` now builds clean (no paged-kv-manager.cpp in the
|
||||
checkout) - proves the split.
|
||||
- Load a published NVFP4 GGUF via the gallery entry, hit /v1/chat/completions, confirm the
|
||||
server log shows LLAMA_KV_PAGED engaged (LLAMA_KV_PAGED_DEBUG trace) and the configured
|
||||
max_batch_tokens/parallel took effect.
|
||||
- go test ./core/gallery/importers/... green (importer drop-in case).
|
||||
- node scripts/changed-backends.js dry-run: editing backend/cpp/llama-cpp/* retriggers
|
||||
llama-cpp-paged (cross-trigger), editing backend/cpp/llama-cpp-paged/* triggers it too.
|
||||
|
||||
================================================================================
|
||||
END OF PLAN
|
||||
================================================================================
|
||||
@@ -1,75 +0,0 @@
|
||||
# Paged bit-exactness gate - per path (canonical references)
|
||||
|
||||
## TL;DR
|
||||
|
||||
The greedy decode of the **paged** path does not byte-match the **non-paged**
|
||||
path for the MoE model. This is a **benign FP-accumulation-order difference of
|
||||
the paged attention reduction**, KL-validated against the f16 reference. It is
|
||||
**not a bug**. The bit-exactness gate is therefore **per path**:
|
||||
|
||||
| path | model | canonical md5 |
|
||||
|------|-------|---------------|
|
||||
| non-paged | MoE q36-35b-a3b-nvfp4 | `07db32c2bcb78d17a43ed18bc22705cd` |
|
||||
| paged | MoE q36-35b-a3b-nvfp4 | `8cb0ce23777bf55f92f63d0292c756b0` |
|
||||
| non-paged | dense q36-27b-nvfp4 | `5951a5b4d624ce891e22ab5fca9bc439` |
|
||||
| paged | dense q36-27b-nvfp4 | `5951a5b4d624ce891e22ab5fca9bc439` (bit-exact to non-paged) |
|
||||
|
||||
Gate command (chat-template / conversation path):
|
||||
```
|
||||
llama-completion -m MODEL -ngl 99 -fa on -p "The capital of France is" \
|
||||
-n 48 --temp 0 --seed 1
|
||||
# paged: prefix with LLAMA_KV_PAGED=1 LLAMA_MOE_FORCE_GRAPHS=1
|
||||
```
|
||||
Note: use the default chat-template path (do **not** pass `-no-cnv`; raw
|
||||
completion lands in a different md5 namespace).
|
||||
|
||||
**Future paged-MoE regressions compare to the PAGED reference `8cb0ce23`, not to
|
||||
the non-paged `07db32c2`.** Dense is bit-exact across paths, so dense uses the
|
||||
single reference `5951a5b4`.
|
||||
|
||||
## Why dense is bit-exact but MoE is not
|
||||
|
||||
Dense paged decode reproduces the non-paged reduction order exactly, so dense
|
||||
greedy md5 is identical across paths. The MoE path runs additional kernels (the
|
||||
NVFP4 MoE GEMM + expert routing) whose multi-kernel accumulation order differs
|
||||
between the paged and non-paged attention layouts. Over a long greedy decode this
|
||||
flips a small number of near-tied argmaxes, changing the byte stream. The same
|
||||
divergence is present on the 0028 baseline, with `LLAMA_MOE_FORCE_GRAPHS` on or
|
||||
off, and with the patch-0029 block-table cache on or off - it is a property of
|
||||
the paged attention path, not of any one lever.
|
||||
|
||||
## KL evidence that the paged path is sound (the load-bearing check)
|
||||
|
||||
`llama-perplexity --kl-divergence` on `q36-35b-a3b-nvfp4.gguf`, 16 chunks,
|
||||
`-c 512 -ngl 99 --seed 1`, base logits from the f16 reference
|
||||
(`darwin_36b_opus/f16.gguf`, PPL 7.3734):
|
||||
|
||||
| comparison | PPL(Q) | KL divergence | Same top p | Cor |
|
||||
|------------|-------:|--------------:|-----------:|----:|
|
||||
| f16 reference | 7.3734 | - | - | - |
|
||||
| **non-paged** vs f16 | 7.3896 | 0.136597 +/- 0.003157 | 84.314% | 97.68% |
|
||||
| **paged** vs f16 | 7.4009 | 0.136000 +/- 0.003285 | 84.828% | 97.58% |
|
||||
| paged vs non-paged (direct) | 7.4009 (base 7.3818) | 0.050011 +/- 0.001653 | 89.044% | 99.04% |
|
||||
|
||||
Direct paged-vs-non-paged: Mean Delta-p = 0.079% (no bias), RMS Delta-p = 6.187%.
|
||||
|
||||
### Verdict: BENIGN
|
||||
|
||||
- **Paged does not diverge from the f16 ground truth more than non-paged does.**
|
||||
KLD(paged||f16) = 0.13600 <= KLD(nonpaged||f16) = 0.13660, and PPL(paged) =
|
||||
7.4009 ~ PPL(nonpaged) = 7.3896 (difference 0.011, far inside the +/- 0.29
|
||||
error bars). A real paged-MoE correctness bug would push paged measurably
|
||||
*further* from f16; it does not (it is marginally closer).
|
||||
- **Paged and non-paged cluster together.** They agree with each other (KLD 0.050,
|
||||
89.0% same-top-p) more than either agrees with f16 (KLD ~0.137, ~84% same-top-p),
|
||||
with essentially zero probability bias. That is the signature of two equivalent
|
||||
FP-reorderings of the same quantized model, both equally approximating the f16
|
||||
ground truth - not a quality regression.
|
||||
- The direct same-top-p of 89.0% is below a naive ">99%" heuristic, but that
|
||||
heuristic is calibrated for higher-precision models. In a 4-bit (NVFP4) model
|
||||
logit near-ties are abundant, so a different-but-equivalent reduction order
|
||||
flips ~11% of argmaxes with no quality cost (proven by the equal KLD-to-f16 and
|
||||
zero Delta-p bias).
|
||||
|
||||
Therefore the canonical gate is per path, and `8cb0ce23` is the validated paged
|
||||
reference for the MoE deployment path.
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,156 +0,0 @@
|
||||
# llama.cpp patch series — paged attention (vLLM-parity engine)
|
||||
|
||||
A **stacking** series: each patch is a small, self-contained, independently-buildable step toward an
|
||||
in-model paged-attention engine. They apply in numeric order on top of the pinned `LLAMA_VERSION`
|
||||
(`backend/cpp/llama-cpp/Makefile`). The build applies them automatically after checkout (see the
|
||||
`llama.cpp:` target). Keeping the work as ordered patches — rather than one big diff — is what lets us
|
||||
**rebase cleanly across llama.cpp bumps and avoid drift**: when a patch stops applying, only that small
|
||||
patch needs fixing, and the failure points at exactly which step the upstream change touched.
|
||||
|
||||
## Base
|
||||
|
||||
- `LLAMA_VERSION` pin in `../Makefile`. **All patches are generated against that exact commit.** Bumping
|
||||
the pin = re-run the regen workflow below and fix only the patches that no longer apply.
|
||||
|
||||
## The series (phases → patches)
|
||||
|
||||
| # | Patch | What | Verifies |
|
||||
|---|-------|------|----------|
|
||||
| 0001 | `0001-vendor-paged-kv-manager.patch` | Add `src/paged-kv-manager.{h,cpp}` (vLLM-parity block manager, CPU foundation) + CMake; no behavior change | builds; unit-tested separately |
|
||||
| 0002 | `0002-paged-kv-storage.patch` | Shared block-pool KV tensor + `set_rows`-by-slot writes, behind `LLAMA_KV_PAGED` | builds; write/gather round-trip |
|
||||
| 0003 | `0003-paged-gather-read.patch` | `build_attn_paged` gather-read in `llama-graph.cpp` | **Gate 0**: token-identical greedy gen, single + multi-seq |
|
||||
| 0004 | `0004-paged-ondemand-alloc.patch` | On-demand block allocation via PagedKVManager | max concurrent seqs before OOM |
|
||||
| 0005 | `0005-paged-continuous-batching.patch` | Block-granular admit/evict in the server slot path | tok/s vs concurrency, mixed-length |
|
||||
| 0006 | `0006-paged-prefix-caching.patch` | Block-hash cross-request prefix dedup | TTFT + memory on shared prefixes |
|
||||
|
||||
Each row is a separate `git commit` on the dev branch (below), exported 1:1 as a patch. Default off
|
||||
(`LLAMA_KV_PAGED`) until Gate 0 (0003) is green, so partial series never changes stock behavior.
|
||||
|
||||
## Regen workflow (the anti-drift recipe)
|
||||
|
||||
```sh
|
||||
# 1. check out the exact pin into a dev tree
|
||||
git -C /tmp clone https://github.com/ggml-org/llama.cpp llama-dev && cd /tmp/llama-dev
|
||||
git checkout <LLAMA_VERSION from ../Makefile>
|
||||
git checkout -b paged
|
||||
|
||||
# 2. apply the current series (each becomes a commit), or develop the next patch
|
||||
git am /path/to/backend/cpp/llama-cpp-localai-paged/patches/paged/00*.patch # or `git apply` + commit per patch
|
||||
|
||||
# 3. iterate a phase as ONE commit, then export the whole series 1:1
|
||||
git format-patch <LLAMA_VERSION>..paged -o /path/to/backend/cpp/llama-cpp-localai-paged/patches/paged/ --zero-commit -N
|
||||
|
||||
# 4. on a pin bump: rebase `paged` onto the new pin; only conflicting patches need edits; re-export.
|
||||
```
|
||||
|
||||
## Build integration
|
||||
|
||||
The series is owned by this backend (`backend/cpp/llama-cpp-localai-paged`), not by the stock
|
||||
`llama-cpp` backend, which is pure upstream. `../Makefile` (the paged wrapper) clones the pinned
|
||||
`llama.cpp` via the copied stock build infra, then applies this series onto the cloned tree with the
|
||||
same strict `git apply` the stock build uses for base patches:
|
||||
```
|
||||
for p in $(PAGED_PATCHES_DIR)/0*.patch; do git apply --verbose "$p" || exit 1; done
|
||||
```
|
||||
All variants (avx/avx2/avx512/cuda/…) clone + apply into their own build copy, so the series ships
|
||||
everywhere without ever touching the stock `llama-cpp` source tree.
|
||||
|
||||
## Latest mirror check
|
||||
|
||||
Phase 37 re-verified the mirror invariant after adding patch `0063`:
|
||||
|
||||
```text
|
||||
base=0ed235ea2c17a19fc8238668653946721ed136fd
|
||||
applied_tree=dedb1182910eafe9f6875588dc8285bfb544cce5
|
||||
fork_tree=dedb1182910eafe9f6875588dc8285bfb544cce5
|
||||
```
|
||||
|
||||
The check used a fresh worktree at `LLAMA_VERSION`, applied every
|
||||
`patches/paged/0*.patch` with strict `git apply`, staged the result, and compared
|
||||
`git write-tree` to canonical fork branch `localai-paged` at
|
||||
`2d590d770 feat(cuda): trace cublas tensor names`.
|
||||
|
||||
Phase 69 re-verified that the committed LocalAI patch series still matches the
|
||||
Phase37 fork tip, and then dry-ran the additive patch export needed for the
|
||||
current local fork HEAD. No generated patch files were edited in Phase69 because
|
||||
the repo policy requires pushing the fork branch before regenerating the LocalAI
|
||||
series, and pushes still require explicit approval.
|
||||
|
||||
Committed-series check:
|
||||
|
||||
```text
|
||||
base=0ed235ea2c17a19fc8238668653946721ed136fd
|
||||
applied_tree=dedb1182910eafe9f6875588dc8285bfb544cce5
|
||||
patch_tip_tree=dedb1182910eafe9f6875588dc8285bfb544cce5
|
||||
fork_head_tree=fcf5720b659c5e1e2b487ccf3c8f7289bb12b9c4
|
||||
match_patch_tip=yes
|
||||
match_fork_head=no
|
||||
patch_count=54
|
||||
```
|
||||
|
||||
Dry-run export from `2d590d770..ea0875d14` produced ten source-only candidate
|
||||
patches:
|
||||
|
||||
```text
|
||||
0064-feat-server-trace-serving-admission-batches.patch
|
||||
0065-feat-server-add-admission-trace-histograms.patch
|
||||
0066-feat-server-add-TTFT-prefill-first-scheduler-mode.patch
|
||||
0067-feat-server-cap-TTFT-prefill-first-decode-deferral.patch
|
||||
0068-feat-server-gate-TTFT-defer-by-prompt-backlog.patch
|
||||
0069-test-cuda-cover-W4A16-direct-activation-policy.patch
|
||||
0070-feat-cuda-route-W4A16-direct-activation-stub.patch
|
||||
0071-feat-cuda-trace-layout-tensor-names.patch
|
||||
0072-feat-cuda-trace-activation-quant-routes.patch
|
||||
0073-feat-cuda-gate-BF16-cuBLAS-F32-output.patch
|
||||
```
|
||||
|
||||
Projected-series check with current `0001..0063` plus temp `0064..0073`:
|
||||
|
||||
```text
|
||||
base=0ed235ea2c17a19fc8238668653946721ed136fd
|
||||
applied_plus_missing_tree=fcf5720b659c5e1e2b487ccf3c8f7289bb12b9c4
|
||||
fork_head_tree=fcf5720b659c5e1e2b487ccf3c8f7289bb12b9c4
|
||||
match_fork_head=yes
|
||||
current_patch_count=54
|
||||
missing_patch_count=10
|
||||
projected_patch_count=64
|
||||
```
|
||||
|
||||
Next mirror action after explicit push approval:
|
||||
|
||||
1. Push `/home/mudler/_git/llama.cpp` branch `localai-paged` to
|
||||
`fork/localai-paged`.
|
||||
2. Regenerate or copy the equivalent source-only `0064..0073` patches from the
|
||||
pushed fork.
|
||||
3. Repeat the projected-series tree hash check above against fork HEAD before
|
||||
committing generated patches.
|
||||
|
||||
## Status
|
||||
|
||||
- **0001 vendor manager — DONE.** Applies clean to the pin; builds into `libllama`.
|
||||
- **0002 block placement — DONE + VERIFIED.** Built `llama-simple` at the pin; greedy generation is
|
||||
**token-identical** stock vs `LLAMA_KV_PAGED=1` (Qwen3-0.6B), paged branch confirmed firing.
|
||||
- **0003 gather-read — DONE + VERIFIED (Gate 0 green).** Implemented in the **additive** form
|
||||
(see `../README.md`): all logic in new `src/paged-attn.{h,cpp}` (a `llm_graph_input_i` gather-index
|
||||
subclass + the K/V/mask gather), hooked by **one** line in `build_attn` + **two** thin accessors on
|
||||
`llama_kv_cache_context` + 1 CMake line (216 insertions; no edit to `llm_graph_input_attn_kv` or
|
||||
`llama-graph.h`). Greedy generation is **token-identical** stock vs `LLAMA_KV_PAGED=1` (Qwen3-0.6B,
|
||||
**9/9** across 3 prompts × {32,96,128} tokens), with `n_gather=71 < n_kv=256` confirming real
|
||||
compaction. Patch: `0003-paged-gather-read-env-LLAMA_KV_PAGED.patch`.
|
||||
- **Key correctness finding:** `get_gather_idxs` must emit cells **sorted by token position**. The CPU
|
||||
flash-attn online softmax reduces cells in physical-array order and is FP-order-sensitive, so 0002's
|
||||
scattered placement *alone* (full-window read, no gather) diverges from stock once a sequence crosses
|
||||
the first 16-cell block. The position-sorted gather reproduces stock's exact reduction order -> bit-
|
||||
identical, not merely mathematically equivalent. So 0002 is the placement substrate; **0003 is what
|
||||
makes paged placement token-identical under flash-attn.**
|
||||
- 0004–0006 follow.
|
||||
|
||||
### Honest parity note (important)
|
||||
|
||||
This series delivers the paged-attention **engine** (capacity + scheduling + prefix sharing). It does **not**
|
||||
by itself reach vLLM throughput parity, because the measured prefill bottleneck is the **FP4 MoE GEMM kernel**
|
||||
(Lever 3: `mul_mat_q<MXFP4>` ~22 TFLOP/s, ~27× behind vLLM) — a *per-token compute* gap that paging does not
|
||||
touch. Paged attention closes the **concurrency/memory** gap (more sequences, prefix reuse); the prefill/throughput
|
||||
gap additionally needs the tcgen05/CUTLASS grouped-GEMM (deferred, upstream-grade, no shortcut — see
|
||||
`../README.md`). So full vLLM parity = this series **AND** the
|
||||
kernel; neither alone suffices.
|
||||
@@ -1,76 +0,0 @@
|
||||
# PREFILL_GEMM_RESULTS - option (a) dequant->bf16 cuBLAS, measured on GB10
|
||||
|
||||
Companion to `PREFILL_GEMM_SCOPE.md`. This records the GPU A/B for the #1
|
||||
prefill lever (route large-M NVFP4 dense GEMMs off FP4-MMQ onto dequant->bf16
|
||||
cuBLAS / nvjet). Shipped as patch `0033`, **default-off** because the measured
|
||||
result is a regression on this hardware.
|
||||
|
||||
Hardware: NVIDIA GB10 (sm_121), CUDA 13.0. Backend pin `9d5d882d`.
|
||||
Models: `q36-27b-nvfp4.gguf` (dense), `q36-35b-a3b-nvfp4.gguf` (MoE).
|
||||
Binary: `build-cuda/bin/llama-batched-bench -fa on -ngl 99`, `LLAMA_KV_PAGED=1`.
|
||||
A/B is a single build toggled by `LLAMA_FP4_PREFILL_M` (0 = MMQ baseline, >0 =
|
||||
route prefill M>threshold to bf16 cuBLAS), so it isolates exactly this lever.
|
||||
|
||||
## 1. Bit-exact / numeric gate (PASS - divergence benign)
|
||||
|
||||
| Gate | Result |
|
||||
|---|---|
|
||||
| `test-backend-ops -o MUL_MAT` (default, threshold off) | 1146/1146 pass |
|
||||
| `test-backend-ops -o MUL_MAT_ID` (default) | 806/806 pass (MoE untouched) |
|
||||
| `test-backend-ops -o MUL_MAT`, path FORCED (`LLAMA_FP4_PREFILL_M=64`) | NVFP4 large-M cases (m=2048/1600/2050, n=128, k=2048) green CUDA-vs-CPU |
|
||||
| greedy md5, short prefill (< threshold), lever vs base | identical: `5951a5b4d624ce891e22ab5fca9bc439` (== documented dense reference; decode byte-untouched) |
|
||||
| greedy md5, long prefill (> threshold, exercises bf16 path), lever vs base | identical: `5f3967df5781445feeb25762abb9eae7` (the new FP path flips no greedy argmax) |
|
||||
|
||||
The new path (NVFP4->bf16 round, bf16 tensor cores, f32 accumulate) is a
|
||||
different FP path from fused FP4xQ8_1 MMQ, but it is precision-neutral-to-better:
|
||||
keeping activations in bf16 instead of Q8_1 is strictly more precise, and the
|
||||
greedy output is byte-identical. This matches the scope's prediction
|
||||
(KLD(dequant-bf16 || f16) <= KLD(FP4-MMQ || f16)).
|
||||
|
||||
## 2. Performance (REGRESSION - the lever loses on GB10)
|
||||
|
||||
S_PP (prefill tokens/s), q36-27b dense, A/B `LLAMA_FP4_PREFILL_M` off vs on:
|
||||
|
||||
| prefill ubatch M | npl | base S_PP (MMQ) | lever S_PP (bf16 cuBLAS) | delta |
|
||||
|---|---|---|---|---|
|
||||
| 512 | 32 | 958.99 | 486.65 | -49% |
|
||||
| 1024 | 8 | 1013.65 | 587.27 | -42% |
|
||||
| 2048 | 8 | 918.46 | 649.42 | -29% |
|
||||
|
||||
Default-off control (no env): S_PP 966.98 == base (within noise) -> the patch is
|
||||
inert by default.
|
||||
|
||||
## 3. Why it loses (the scope premise was wrong for GB10)
|
||||
|
||||
The scope assumed FP4-MMQ is register-bound to ~3% of FP4 peak at large M, so a
|
||||
vendor large-M kernel would win. **Measured, FP4-MMQ at M=512..2048 beats
|
||||
dequant->bf16 cuBLAS by 29-49%.** Two compounding reasons:
|
||||
|
||||
1. **bf16 tensor-core peak is ~half FP4 peak on GB10.** Even a perfect bf16 GEMM
|
||||
caps at ~half the throughput the FP4-MMA path can reach.
|
||||
2. **The dequant tax is an un-amortized memory pass.** Per prefill step the new
|
||||
path reads FP4 weights (~0.5 B/elt), writes bf16 (2 B/elt), then the GEMM
|
||||
reads bf16 (2 B/elt) = ~8x the weight byte traffic of the FP4-MMQ read
|
||||
(~0.5 B/elt). The dequant write is M-independent, so it only amortizes as M
|
||||
grows: the gap shrinks 49% -> 42% -> 29% from M=512 -> 2048 but never crosses
|
||||
even at M=2048 (above the default n_ubatch).
|
||||
|
||||
This is also consistent with the README decode finding that the dense path was
|
||||
already ~96-97% of vLLM - the dense GEMM was never the bottleneck the way the
|
||||
prefill ground-truth (measured on the MoE decision model) implied.
|
||||
|
||||
## 4. Status of the phases
|
||||
|
||||
- **Phase 1 (dense): REJECTED on GB10**, landed default-off as a validated,
|
||||
env-gated scaffold (mechanism + bit-exact gate reusable by option (b) and by
|
||||
non-GB10 hardware where bf16 may fare differently).
|
||||
- **Phase 2 (MoE grouped large-M): NOT implemented.** It inherits the same
|
||||
bf16-peak < FP4-peak ceiling plus a per-expert dequant, so a grouped
|
||||
bf16-cuBLAS would regress for the same reason; the MoE id-path also has the
|
||||
graph-safety catch (a false `should_use_mmq` falls to the host-sync sorted
|
||||
loop, not CUDA-graph-safe). Not worth the multi-day grouped-cuBLAS + graph
|
||||
work on a path the dense A/B already shows loses.
|
||||
- **The only route to a real prefill GEMM win is option (b)** - a native
|
||||
Blackwell FP4-MMA large-M kernel (multi-week), to greenlight only if the
|
||||
prefill regime is funded. The committed scaffold gives option (b) its
|
||||
M-threshold routing and its bit-exact gate for free.
|
||||
@@ -1,264 +0,0 @@
|
||||
# PREFILL_GEMM_SCOPE - large-M NVFP4 expert/dense GEMM (design only)
|
||||
|
||||
**Status: DESIGN + PLAN ONLY. No kernel written, no GPU run in this pass.**
|
||||
This scopes the #1 prefill lever for `llama-cpp-localai-paged`: the NVFP4 weight
|
||||
GEMM at large M (prefill), where llama.cpp's `mul_mat_q` (MMQ) NVFP4 path is far
|
||||
slower than vLLM's `marlin_moe_wna16` (MoE) + cutlass/nvjet (dense). Per the
|
||||
prefill ground-truth that motivated this scope, the GEMM bucket is ~232 us/tok
|
||||
(paged) vs ~68 us/tok (vLLM) - 3.4x slower, ~51% of the paged-vs-vLLM prefill
|
||||
gap (164 us/tok).
|
||||
|
||||
> **Regime warning (read first).** Every "GEMM is at the BW floor / ties vLLM"
|
||||
> conclusion in `README.md` section 5 is a **DECODE** finding (M<=128,
|
||||
> bandwidth-bound). This document is about **PREFILL** (large M, compute /
|
||||
> tensor-core-throughput bound) - a different regime, which is exactly why the
|
||||
> rejected "W4A16-Marlin MoE GEMM" lever is revisited here **for prefill only**.
|
||||
> The 232/164/68 us/tok prefill bucket came from the prefill ground-truth that
|
||||
> commissioned this scope and is **not** in a committed in-repo profile (the
|
||||
> committed profiling - `GAP_PROGRESS.md` etc. - is decode-focused). Per the
|
||||
> "profile-don't-assume" rule in `.agents/vllm-parity-methodology.md`, **step 0 of
|
||||
> any build is to re-confirm the prefill GEMM bucket on GPU** (nsys, prefill-only
|
||||
> window) before touching code.
|
||||
|
||||
---
|
||||
|
||||
## 1. Why `mul_mat_q` is slow at large M (confirmed from source)
|
||||
|
||||
Source: `ggml/src/ggml-cuda/mmq.cu`, `mmq.cuh` at this backend's pin (`9d5d882d`).
|
||||
|
||||
MMQ is built for the **M<=128 decode tile**. Three structural facts from the code:
|
||||
|
||||
1. **The M (column/token) tile is capped at 128.**
|
||||
`get_mmq_x_max_host()` / `get_mmq_x_max_device()` (mmq.cuh ~108-140) return
|
||||
`128` on Blackwell (`turing_mma_available(cc)`), and the host launch loop
|
||||
(mmq.cuh ~4237) picks `mmq_x_best` only to *minimise the column-tile count for
|
||||
`ncols_max`, never exceeding `mmq_x_max`*. So a prefill ubatch of M=512 (or
|
||||
4096) tokens is processed as many `mmq_x<=128` column-tiles. The compile-time
|
||||
accumulator tile is `mmq_x`-wide; there is no large-M (e.g. 256-wide) tile
|
||||
variant. The whole tile-selection machinery exists to pick a *small* tile for
|
||||
*small* batches, not to grow for large ones.
|
||||
|
||||
2. **The FP4-MMA kernel is register-bound to 1 CTA/SM.**
|
||||
`mul_mat_q` for FP4 is `__launch_bounds__(warp_size*nwarps, min_blocks=1)`
|
||||
(mmq.cuh ~3579-3585), i.e. 256 threads, 1 resident block/SM (~255 regs/thread).
|
||||
The patch-0017 comment in-tree states this plainly: the kernel is
|
||||
"REGISTER-bound to 1 CTA/SM ... the under-occupancy that strands the kernel at
|
||||
~3% of FP4 peak at M=128." At large M the work per tile is bigger, but with one
|
||||
CTA/SM the tensor cores still stall on LPDDR5x / shared-memory weight loads
|
||||
with no CTA-level latency hiding - the design has no async multi-stage global->
|
||||
shared pipeline (cp.async double-buffering) that large-M GEMMs need.
|
||||
|
||||
3. **Per-tile fixed overheads amortise poorly only because the tile stays small.**
|
||||
Each tile re-stages weights into shared memory, runs the `MMQ_ITER_K_FP4=512`
|
||||
K-loop, and the activations are quantized to Q8_1 (`quantize_mmq_fp4_cuda`,
|
||||
block_fp4_mmq = FP4 weights x int8 activations). For decode this is the right
|
||||
trade (FP4 weight traffic is the bottleneck). For large-M prefill the GEMM is
|
||||
compute-bound, so the right structure is big tensor-core output tiles (e.g.
|
||||
128x256), a deep async load pipeline, and full SM occupancy - exactly what
|
||||
cutlass 3.x / nvjet (cuBLAS) and marlin implement and MMQ does not.
|
||||
|
||||
Patch 0017 already proved every *cheap* large-tile/occupancy lever inside MMQ
|
||||
(`GGML_CUDA_FP4_MMQ_Y`, `GGML_CUDA_FP4_MINBLOCKS`) is a no-win on GB10 - because
|
||||
the limit is the small-tile kernel *structure*, not a tunable. To win at large M
|
||||
you must leave MMQ for a large-M kernel.
|
||||
|
||||
---
|
||||
|
||||
## 2. Options (feasibility / bit-exactness / effort)
|
||||
|
||||
### Key enabling facts already in the tree
|
||||
|
||||
- **NVFP4 -> bf16/f16 dequant kernels already exist.** `convert.cu` defines
|
||||
`dequantize_row_nvfp4_cuda`; `ggml_get_to_bf16_cuda` / `ggml_get_to_fp16_cuda`
|
||||
/ `ggml_get_to_fp16_nc_cuda` all return it for `GGML_TYPE_NVFP4`. The
|
||||
non-Blackwell fallback ("falls back to dequant", README s2) already uses this.
|
||||
- **cuBLAS on GB10 dispatches to nvjet** (NVIDIA's JIT tensor-core GEMM) - the
|
||||
committed profiles already show `nvjet lm_head` and `nvjet non-FP4 cublas GEMM`
|
||||
rows. So a dequant->cuBLAS bf16 GEMM lands on a vendor-tuned large-M kernel for
|
||||
free.
|
||||
- **BUT NVFP4 is explicitly excluded from the tensor-core cuBLAS path.** In
|
||||
`ggml_cuda_op_mul_mat_cublas` (ggml-cuda.cu ~1659) the `use_fp16` predicate
|
||||
begins `src0->type != GGML_TYPE_NVFP4 && ...`. So if NVFP4 reaches cuBLAS today
|
||||
it falls to the `else` branch: dequant to **F32** + `cublasSgemm` (**no tensor
|
||||
cores**) - useless for prefill. Relaxing this one exclusion (route NVFP4 to the
|
||||
bf16/f16 tensor-core branch, where `to_*_cuda(NVFP4)` already exists) is the
|
||||
pivot that makes option (a) a few-line change rather than a kernel.
|
||||
|
||||
### (a) Dequant -> cuBLAS/cutlass bf16 GEMM for large M -- RECOMMENDED
|
||||
|
||||
Dequant the NVFP4 weights to bf16 (transient pool buffer) once per prefill step,
|
||||
then a large-M tensor-core `cublasGemmEx` (CUBLAS_COMPUTE_32F accumulate, bf16
|
||||
inputs). Activations stay bf16 (not Q8_1-quantized).
|
||||
|
||||
- **Feasibility: HIGH.** All pieces exist (dequant kernels, cuBLAS bf16 path,
|
||||
pool allocator). The only code change for the dense path is (i) make
|
||||
`ggml_cuda_should_use_mmq` return false for NVFP4 dense above an M threshold so
|
||||
the dispatch falls through to `ggml_cuda_op_mul_mat_cublas`, and (ii) relax the
|
||||
`src0->type != GGML_TYPE_NVFP4` exclusion so it dequants to bf16 and uses
|
||||
`cublasGemmEx` tensor-core, not f32 Sgemm.
|
||||
- **Cost model (the crux - why it wins ONLY at large M).** Dequant is one extra
|
||||
weight-sized memory pass (read ~0.5B/elt FP4 + scales, write 2B/elt bf16). The
|
||||
bf16 GEMM then reads weights as bf16 = **4x the byte traffic of the FP4-MMQ
|
||||
read**. At small M (decode) this 4x weight traffic dominates -> bf16-cuBLAS
|
||||
loses -> keep MMQ (this is why decode stays FP4-MMQ; consistent with the
|
||||
README decode verdict). At large M the GEMM is compute-bound and weight traffic
|
||||
is amortised over hundreds of columns, so the 4x is cheap and cuBLAS's mature
|
||||
large tiles + async pipeline + full occupancy dominate MMQ's 3%-of-peak small
|
||||
tile. The dequant pass itself is ~one weight-read amortised over the whole
|
||||
prefill step - negligible at large M.
|
||||
- **Honest ceiling.** GB10 bf16 tensor-core peak is ~**half** the FP4 tensor-core
|
||||
peak. A bf16 cuBLAS GEMM at ~70-80% of bf16 peak is ~35-40% of FP4 peak. That
|
||||
is a huge jump from MMQ's ~3% large-M utilisation, but it is **not** automatic
|
||||
full vLLM parity (vLLM prefill uses 4-bit weight tiles, staying near FP4-class
|
||||
throughput). Expect this to recover most, not all, of the 232->68 gap. See s4.
|
||||
- **Bit-exactness: NEW FP path** (NVFP4->bf16 round, bf16 TC, f32 accumulate) vs
|
||||
fused FP4xQ8_1 MMQ. **Not byte-identical** - gate per-path via KLD exactly like
|
||||
the paged-MoE `8cb0ce23` precedent (README s5 / `PAGED_BITEXACT_NOTE.md`). It
|
||||
should pass *easily and favourably*: keeping activations in bf16 instead of
|
||||
Q8_1 is strictly more precise than the MMQ path, so KLD(dequant-bf16 || f16)
|
||||
should be <= KLD(FP4-MMQ || f16). This is a precision-neutral-to-better change,
|
||||
not a precision regression like the rejected lever 4.
|
||||
- **Effort: LOW-MEDIUM (a few days).** Dispatch flip + exclusion relax + an M
|
||||
threshold + the KL gate + a prefill bench. No new kernel. Dense first; MoE is
|
||||
the harder follow-on (see (c)/plan).
|
||||
- **Memory note.** Dequant into a *transient* pool scratch per step (do **not**
|
||||
cache bf16 weights - a persistent bf16 copy is 4x VRAM for those tensors and
|
||||
would erase the backend's "1.5-3x less memory" property). The per-step dequant
|
||||
pass is the price of keeping the model FP4-resident.
|
||||
|
||||
### (b) Marlin-style fused NVFP4 large-M MoE GEMM (port `marlin_moe_wna16`)
|
||||
|
||||
Port vLLM's marlin grouped MoE kernel (4-bit weights, f16 activations, dequant-
|
||||
in-register, async cp.async pipelines, swizzled layouts).
|
||||
|
||||
- **Feasibility: LOW (hardest).** Marlin is a hand-tuned CUTLASS-class kernel and
|
||||
is **not NVFP4-aware** (it targets wna16 group-quant, not NVFP4's 16-elt blocks
|
||||
with ue4m3 micro-scales). You would either (i) adapt marlin to dequant NVFP4
|
||||
in-register and accumulate in f16 (abandoning native Blackwell FP4-MMA), or
|
||||
(ii) write a brand-new Blackwell sm_121 FP4-MMA large-M kernel - which is
|
||||
essentially re-implementing what cutlass 3.x / nvjet already give you via (a).
|
||||
- **Bit-exactness:** new FP path, KL-gate (same as (a)).
|
||||
- **Effort: HIGH (multi-week, high risk),** kernel + layout + Blackwell MMA
|
||||
scheduling + graph-safety + the bit-exact gate.
|
||||
- **Verdict: do NOT start here.** Its only structural advantage over (a) is 4-bit
|
||||
weight traffic, which matters only when BW-bound = small M = **decode**, the
|
||||
regime already rejected. At large M (a) reaches the same vendor large-M kernels
|
||||
for ~1% of the effort. Keep (b) on the shelf as the *only* route to true 68
|
||||
us/tok parity if (a)'s bf16 ceiling proves insufficient and the win justifies a
|
||||
multi-week kernel.
|
||||
|
||||
### (c) M-threshold routing (the integration mechanism for (a))
|
||||
|
||||
Not an alternative to (a) - it is *how* (a) is wired. Keep FP4-MMQ for decode
|
||||
(M<=threshold), switch to the large-M path for prefill.
|
||||
|
||||
- **Cleanest hook:** `ggml_cuda_should_use_mmq(type, cc, ne11_or_ne12, n_experts)`
|
||||
already receives M (`ne11` dense / `ne12` MoE tokens). Add an NVFP4+Blackwell
|
||||
branch: return false when M > `LLAMA_FP4_PREFILL_M` (default e.g. 256-512,
|
||||
env/`-D` tunable, default value chosen so default == today's behaviour until
|
||||
validated). It is called from both `ggml_cuda_mul_mat` (~2573/2582) and
|
||||
`ggml_cuda_mul_mat_id` (~2664), so one edit covers dense + MoE routing.
|
||||
- **Dense fallthrough is clean:** `ggml_cuda_mul_mat` final `else` ->
|
||||
`ggml_cuda_op_mul_mat(..., ggml_cuda_op_mul_mat_cublas, ...)` -> with the
|
||||
exclusion relaxed, dequant->bf16->`cublasGemmEx`. Works.
|
||||
- **MoE fallthrough is NOT clean (the catch):** in `ggml_cuda_mul_mat_id`, a
|
||||
false `should_use_mmq` falls to `should_use_mmf` (no NVFP4 support) then to the
|
||||
**host-side sorted per-expert loop** with a `cudaStreamSynchronize` (ggml-cuda.cu
|
||||
~2700) - slow and **not CUDA-graph-safe** (it would break the MoE re-graph,
|
||||
patch 0025). So MoE large-M needs a *dedicated graph-safe grouped GEMM* (dequant
|
||||
the expert-gathered weights to bf16 + `cublasGemmGroupedBatchedEx`, CUDA 12.5+,
|
||||
over the existing `expert_bounds`/`ids_dst` sorted layout), not a bare
|
||||
fallthrough. This is why the plan ships **dense first, MoE second**.
|
||||
|
||||
---
|
||||
|
||||
## 3. Recommended approach + implementation plan
|
||||
|
||||
**Recommendation: (a) dequant->bf16 cuBLAS, wired via (c) M-threshold routing,
|
||||
dense-path first, MoE grouped-cuBLAS second. Reject (b).**
|
||||
|
||||
### Phase 0 - confirm the bucket on GPU (no code)
|
||||
- nsys prefill-only window (`-npp <large> -ntg 0/1`, exclude the graph-capture
|
||||
step) on q36-27b dense and q36-35b-a3b MoE at the backend pin. Confirm the
|
||||
NVFP4 `mul_mat_q` / `mul_mat_id` bucket is ~232 us/tok and that it is
|
||||
compute-bound at prefill M (check tensor-core active % low, not BW-saturated).
|
||||
If the bucket is not what the ground-truth claims, stop and re-scope.
|
||||
|
||||
### Phase 1 - dense large-M NVFP4 -> bf16 cuBLAS (the bankable win)
|
||||
Files / edits:
|
||||
1. `ggml/src/ggml-cuda/mmq.cu` - `ggml_cuda_should_use_mmq`: add
|
||||
`if (type==GGML_TYPE_NVFP4 && blackwell_mma_available(cc) && ne11 > LLAMA_FP4_PREFILL_M && n_experts==0) return false;`
|
||||
(n_experts==0 = dense only in Phase 1). Default threshold == effectively
|
||||
disabled until A/B-validated, env/`-D` overridable (mirror the 0017
|
||||
`GGML_CUDA_FP4_*` knob style + in-tree comment).
|
||||
2. `ggml/src/ggml-cuda/ggml-cuda.cu` - `ggml_cuda_op_mul_mat_cublas`: relax the
|
||||
`src0->type != GGML_TYPE_NVFP4` guard in `use_fp16` (prefer a dedicated bf16
|
||||
branch: NVFP4 -> `ggml_get_to_bf16_cuda` -> `cublasGemmEx` CUDA_R_16BF /
|
||||
COMPUTE_32F, matching the existing BF16 src0 branch for best accuracy).
|
||||
3. Transient pool scratch for the dequanted weights (reuse `ggml_cuda_pool_alloc`
|
||||
as the existing branch does; no persistent allocation).
|
||||
|
||||
### Phase 2 - MoE grouped large-M (the harder, higher-value follow-on)
|
||||
1. New grouped path reached from `ggml_cuda_mul_mat_id` when
|
||||
`should_use_mmq`==false for NVFP4+large-M+`n_experts>0`: dequant the
|
||||
expert-gathered weights to bf16 and run `cublasGemmGroupedBatchedEx` over the
|
||||
existing `expert_bounds` / `ids_dst` sorted layout that `mul_mat_q` already
|
||||
builds. Reuse the patch-0023 de-dup'd activation gather where applicable.
|
||||
2. **Must stay CUDA-graph-safe** - no host sync (do not fall into the legacy
|
||||
sorted loop). Validate the MoE re-graph (patch 0025 / `LLAMA_MOE_FORCE_GRAPHS`)
|
||||
still captures.
|
||||
|
||||
### The bit-exact / KL gate (both phases)
|
||||
- Greedy md5 on the standard prompt (README s5) to detect *unexpected* divergence
|
||||
on the non-prefill paths (must stay == the per-path reference: dense
|
||||
`5951a5b4`, paged-MoE `8cb0ce23`). The large-M path itself will differ -> gate
|
||||
it by KLD vs the f16 reference, requiring `KLD(new||f16) <= KLD(FP4-MMQ||f16)`
|
||||
and PPL within the established band, recorded in `PAGED_BITEXACT_NOTE.md`.
|
||||
- `test-backend-ops` MUL_MAT / MUL_MAT_ID at NVFP4 **prefill shapes** (large M)
|
||||
CUDA0-vs-CPU, plus the existing decode shapes to prove decode is byte-untouched
|
||||
(default threshold keeps decode on MMQ).
|
||||
|
||||
### The bench
|
||||
- `llama-batched-bench -fa on -ngl 99` reporting **S_PP** (prefill t/s), swept
|
||||
over prefill length and `npl`, A/B with `LLAMA_FP4_PREFILL_M` off vs on, dense
|
||||
and MoE, vs stock and vs the vLLM prefill reference. Per-lever A/B discipline
|
||||
(`.agents/vllm-parity-methodology.md`): one knob at a time, record the rejected
|
||||
threshold values too.
|
||||
|
||||
---
|
||||
|
||||
## 4. Honest risk + expected speedup
|
||||
|
||||
- **Phase 1 (dense) is a tractable routing change, not a kernel project** - days,
|
||||
low risk. It reuses existing dequant kernels and the existing nvjet/cuBLAS
|
||||
large-M path; the net new code is a threshold + a one-line exclusion relax + a
|
||||
KL gate.
|
||||
- **Phase 2 (MoE) is medium risk** - the grouped-batched cuBLAS wiring +
|
||||
CUDA-graph-safety is real work (the bare fallthrough is a slow, graph-breaking
|
||||
host loop), but still far short of a from-scratch kernel.
|
||||
- **Will the GEMM bucket hit 232 -> ~68 us/tok (full vLLM parity)? Honestly, no -
|
||||
not from bf16-cuBLAS alone.** bf16 tensor-core peak on GB10 is ~half FP4 peak,
|
||||
so the realistic floor for a dequant->bf16 GEMM is ~**90-130 us/tok** (roughly
|
||||
35-45% of FP4 peak at ~70-80% of bf16 peak). That recovers ~**60-75%** of the
|
||||
232->68 bucket gap = a large prefill win (the GEMM is ~51% of the total prefill
|
||||
gap, so closing ~two-thirds of it is a meaningful S_PP improvement), but it
|
||||
leaves a residual. **True 68 us/tok parity requires a native FP4-MMA large-M
|
||||
kernel (option (b)) - the multi-week project** to greenlight only if Phase 1's
|
||||
measured win proves the prefill regime matters enough to fund it.
|
||||
- **Recommendation:** build Phase 1, measure, and let the measured dense S_PP
|
||||
gain decide whether Phase 2 (MoE grouped cuBLAS) and ultimately (b) (native FP4
|
||||
large-M kernel) are worth funding. Bank the cheap two-thirds before paying for
|
||||
the kernel.
|
||||
|
||||
---
|
||||
|
||||
## 5. Summary table
|
||||
|
||||
| Option | Feasibility | Bit-exact | Effort | Verdict |
|
||||
|---|---|---|---|---|
|
||||
| (a) dequant->bf16 cuBLAS large-M | HIGH (parts exist) | new FP path, KL-gate (likely better PPL) | LOW-MED (days) | **RECOMMENDED** (dense first) |
|
||||
| (b) Marlin/native FP4 large-M kernel | LOW | new FP path, KL-gate | HIGH (multi-week) | shelf - only route to true 68 us/tok |
|
||||
| (c) M-threshold routing | HIGH | n/a (mechanism) | LOW | **the wiring for (a)** |
|
||||
|
||||
Decode is untouched by all of the above (threshold keeps M<=128 on FP4-MMQ); this
|
||||
is a **prefill-only** lever.
|
||||
@@ -1,628 +0,0 @@
|
||||
# Tensor-Core GDN Build Plan
|
||||
|
||||
> Auto-generated from the GDN build-design workflow. Build-ready spec for the full tensor-core chunked-scan kernel (2nd prefill lever).
|
||||
|
||||
## 1. Remaining intra-chunk products -> mma mapping
|
||||
|
||||
I have everything needed: the exact chunked-scan math from patch 0031, the sm_121a constraints from the scope doc, and the concrete ggml tf32 fragment (`mma.sync.aligned.m16n8k8.row.col.f32.tf32.tf32.f32`, `tile<16,8,float> D, tile<16,8,float> A, tile<8,8,float> B`) at `mma.cuh:976-984`. Here is the design.
|
||||
|
||||
---
|
||||
|
||||
# Tensor-core mapping of the REMAINING intra-chunk GDN products (patch 0031 steps 3-7)
|
||||
|
||||
## 0. Building block + what the PoC already covered
|
||||
|
||||
**Grounding.** Math: `backend/cpp/llama-cpp-localai-paged/patches/paged/0031-paged-chunked-gdn-prefill-scan-kernel.patch` (steps reproduced inline below). Scope/constraints: `backend/cpp/llama-cpp-localai-paged/docs/TENSORCORE_GDN_SCOPE.md`. Fragment API: `ggml/src/ggml-cuda/mma.cuh:976-984` (the only f32-accumulate tf32 overload on sm_121a).
|
||||
|
||||
The single warp-level primitive on sm_121a is **`m16n8k8` tf32 / f32-accumulate**:
|
||||
- `A` fragment = `tile<16,8,float>` (M=16, K=8; 4 floats/thread, `Axi[0..3]`)
|
||||
- `B` fragment = `tile<8,8,float>` (K=8, N=8, `.col` operand; 2 floats/thread, `Bxi[0..1]`)
|
||||
- `D` accumulator = `tile<16,8,float>` (M=16, N=8; 4 floats/thread)
|
||||
- A GEMM `[M×K]·[K×N]` tiles to `ceil(M/16) × ceil(N/8) × ceil(K/8)` mma calls, f32-accumulating over the K-subtiles.
|
||||
- bf16 alternative `m16n8k16` (`mma.cuh:1064`, K=16/mma, 7-bit mantissa) exists but is **only** admissible for the tf32-safe Gram class — never the state/decay-coupled class.
|
||||
- 3xtf32 ladder = split each f32 operand into 3 tf32 limbs, run 3 limb-products per K-subtile (hi·hi, hi·lo, lo·hi), accumulate high→low. ~3x the mma count, ~f32 accuracy.
|
||||
|
||||
**PoC covered products 1 + 2** (the two `C×C` Gram products, both tf32-safe, NMSE ~3e-9): `KK[t,t']=k_t·k_t'` → `A`, and `QK[t,t']=q_t·k_t'` → `P`. Both are `(C×dk)·(dk×C)`, M=C N=C K=dk=128, decay+beta applied in f32 after. They already share the `Kc^T` B-fragments.
|
||||
|
||||
The remaining families are **steps 3,4,5,6,7**. Notation: `C` = chunk (default 64; PoC 16), `dk=dv=128`, per `(head,seq)` block. Tile counts below are for **C=64**.
|
||||
|
||||
---
|
||||
|
||||
## 1. Per-product mma mapping table (the deliverable)
|
||||
|
||||
| # | Product (0031 step) | Result = matmul | M | N | K | mma tiles `(M/16)·(N/8)·(K/8)` @C=64 | Accumulation order | Precision class | Shares staged operand with |
|
||||
|---|---|---|---|---|---|---|---|---|---|
|
||||
| 1 | `KK→A` (PoC) | `Kc · Kcᵀ` | C | C | dk=128 | 4·8·16 = **512** (~½ tri) | over 16 k-subtiles | **tf32-safe** (proven) | `Kcᵀ` B-frag ↔ P2; `Kc` LHS ↔ P3 |
|
||||
| 2 | `QK→P` (PoC) | `Qc · Kcᵀ` | C | C | dk=128 | 4·8·16 = **512** (~½ tri) | over 16 k-subtiles | **tf32-safe** (proven) | `Kcᵀ` B-frag ↔ P1; `Qc` LHS ↔ P4 |
|
||||
| 3 | `KS = S0ᵀk_t` | `Kc · S0` | C | dv=128 | dk=128 | 4·16·16 = **1024** | 16 k-subtiles, limbs hi→lo | **3xtf32 / f32** (state-boundary, feeds solve) | `S0` B-frag ↔ P4; `Kc` LHS ↔ P1 |
|
||||
| 4 | `QS = S0ᵀq_t` | `Qc · S0` | C | dv=128 | dk=128 | 4·16·16 = **1024** | 16 k-subtiles, limbs hi→lo | **3xtf32 → demote-first** (×γ_t≤1 attenuated) | `S0` B-frag ↔ P3; `Qc` LHS ↔ P2 |
|
||||
| 5 | `O += P·U` | `P · U` | C | dv=128 | C=64 | 4·16·8 = **512** (~½ tri over K) | C/8 k-subtiles, triangular | **tf32-safe** (P decay-masked & bounded in f32 first) | `P`(=Amat) ↔ P2; `U` B-frag ↔ P6 |
|
||||
| 6 | `S_C += Kᵀ(D·U)` | `Kcᵀ · DU` | dk=128 | dv=128 | C=64 | 8·16·8 = **1024** | scale state by γ_last (f32) **first**, then C/8 k-subtiles, limbs hi→lo | **3xtf32 / f32** (THE cross-chunk carry, compounds over n_tok/C) | `U` B-frag ↔ P5; `Kc` (transposed) ↔ P1/3 |
|
||||
| 7 | `U = A⁻¹·RHS` off-diag coupling `A_ij·U_j` | `A_ij · U_j` | b=16 | dv=128 | b=16 | 1·16·2 = **32**/pair → **~192** (6 pairs) +~128 diag | forward sweep i=0..C/b; off-diag subtractions before diagonal solve | **tf32-safe off-diag + f32 in-register `16×16` diagonal** | `A`(=Amat) ↔ P1; `U` blocks ↔ P5/P6 |
|
||||
|
||||
3xtf32 inflation if the ladder is taken: P3 1024→**3072**, P4→**3072**, P6 1024→**3072**.
|
||||
|
||||
---
|
||||
|
||||
## 2. Per-product detail (the 5 remaining families)
|
||||
|
||||
### Product 3 - `KS = S0ᵀ k_t` (RHS state-boundary term)
|
||||
0031: `ks = Σ_i Sd[j·dk+i]·Kc[t·dk+i]`; feeds `RHS[t][j] = β_t(v_t[j] − γ_t·ks)`.
|
||||
- **As a GEMM:** `KS[t][j] = Σ_i Kc[t][i]·S0[i][j]` ⇒ `KS = Kc[C×dk] · S0[dk×dv]`. **M=C, N=dv=128, K=dk=128.** Contraction over the state-row index `i`.
|
||||
- **Schedule:** `Kc` is the LHS (M-major over `t`, K over `i`) — already staged for P1. `S0` is the B operand, K-major over `i`, N over `j`. The patch's `Sd[j·dk+i]` layout (i contiguous for fixed j) **is already a K-major B layout** → `ldmatrix`-friendly as `tile<8,8>` B fragments. Accumulate 16 k-subtiles into f32 D.
|
||||
- **Precision: 3xtf32/f32.** This is a state-boundary product: `S0` carries the full sequence history (wide dynamic range), and the result is *differenced* against `v_t` then fed into the solve, so error here propagates through `U` into both `O` and `S_C`. Default to the 3xtf32 ladder; A/B a plain-tf32 demote only after P4.
|
||||
|
||||
### Product 4 - `QS = S0ᵀ q_t` (γ cross-chunk `O` term)
|
||||
0031: `qs = Σ_i Sd[j·dk+i]·Qc[t·dk+i]`; `o = γ_t·qs + Σ P·U`.
|
||||
- **As a GEMM:** `QS = Qc[C×dk] · S0[dk×dv]`. **M=C, N=dv=128, K=dk=128** — identical shape to P3.
|
||||
- **Schedule:** identical to P3 but LHS=`Qc` (shared with P2). **Fuse with P3 on the shared `S0` B-fragments:** stage `S0` once as B, run `Kc·S0` then `Qc·S0` back-to-back — `S0` is the heavy operand (128×128) and is loaded once for both.
|
||||
- **Precision: 3xtf32 but the demote-first candidate.** The term is scaled by `γ_t ≤ 1` in f32 after the mma, so when the chunk has decayed (`γ_t→0`) the absolute error is attenuated. Least sensitive of the three state-boundary products; it is the first to try at plain tf32 in the precision A/B.
|
||||
|
||||
### Product 5 - `O += P · U` (attention-weighted output)
|
||||
0031: `o += Amat[t·Cc+tp]·Ud[j·C+tp]` for `tp≤t`.
|
||||
- **As a GEMM:** `O[C×dv] += P[C×C] · U[C×dv]`. **M=C, N=dv=128, K=C=64.** Contraction over the chunk index `t'`.
|
||||
- **Schedule:** `P` (=Amat scratch from P2, with `d(t',t)` applied in f32) is LHS (M over t, K over t'); `U` (solved, in `Ud`) is the B operand, K-major over t'. `P` is **lower-triangular** ⇒ for M-tile `m` only K-subtiles `≤ m` are non-zero → ~½ the mma. Accumulate `C/8` k-subtiles. Add the `γ_t·QS` term (P4) into the same f32 D accumulator before write-out.
|
||||
- **Precision: tf32-safe.** `P = d·QK` with `d≤1` is formed and bounded **in f32 first** (strong-decay rows already underflowed to ~0), so down-casting the bounded `P` to tf32 for this mma is benign. The decay is never inside the accumulation — it is pre-baked in f32, preserving the bounded de-gating invariant.
|
||||
|
||||
### Product 6 - `S_C += Kᵀ(D·U)` (the state update)
|
||||
0031: `s = γ_last·Sd[j·dk+i] + Σ_t d(t,last)·Kc[t·dk+i]·Ud[j·C+t]`.
|
||||
- **As a GEMM:** let `DU[t][j] = d(t,last)·U[t][j]` (D=diag applied in f32). `S_C[i][j] += Σ_t Kc[t][i]·DU[t][j]` ⇒ `S_C[dk×dv] += Kcᵀ[dk×C] · DU[C×dv]`. **M=dk=128, N=dv=128, K=C=64.** Contraction over the chunk index `t`.
|
||||
- **Schedule:** the accumulator D fragments **are the register-resident state** that persists across the chunk loop. Order is strict: (i) scale the state fragments by `γ_last` in f32 in-register, **then** (ii) mma-accumulate `Kcᵀ·DU` over `C/8` k-subtiles into them. LHS = `Kc` read **transposed** (i as M-row, t as K) — a different fragment view of the same `Kc` smem buffer (use the `ldmatrix` transpose / J-major tile view). B = `DU` = `U` scaled by `d(t,last)` in f32, K-major over t — **same `U` B-layout as P5**.
|
||||
- **Precision: 3xtf32 / f32 — the strongest ladder candidate.** This is the only product whose error *compounds across all `n_tokens/C` chunk steps*; it defines the state trajectory. Keep at 3xtf32 longest; this is the last product to ever consider demoting, and the place where a full-f32 accumulate (3xtf32) is most justified even if everything else passes plain tf32.
|
||||
|
||||
### Product 7 - the A-inverse (blocked forward substitution, FLA UT-transform)
|
||||
0031 does a serial per-thread fwd-subst. Tensor-core form (block `b=16` = one mma M-tile, `C/b=4` blocks at C=64):
|
||||
- For block `i`: `U_i = Ainv_ii·(RHS_i − Σ_{j<i} A_ij·U_j)`.
|
||||
- **The A-inverse-adjacent matmul = the off-diagonal coupling `A_ij·U_j`:** **M=b=16, N=dv=128, K=b=16** ⇒ `1·16·2 = 32` mma/pair; 6 lower pairs at C=64 → **192** mma. Optional materialized-`Ainv_ii` apply is the same shape (~128 more).
|
||||
- **Schedule:** forward sweep `i=0..3`; for each `i` accumulate all `j<i` couplings into a `b×dv` register tile (subtract from `RHS_i`), then apply the `b×b` diagonal inverse. `A`=Amat (from P1, β·d applied in f32) is the LHS; `U_j` blocks are read from `Ud` and updated in place as the sweep advances.
|
||||
- **Precision: split.** Off-diagonal coupling = **tf32-safe** (`A_ij`=β·d·kk is bounded, `d≤1`; well-conditioned for the stable de-gating). The `16×16` **diagonal block inverse stays f32/in-register** (Neumann series on the b-nilpotent, ≤b−1 terms, or a short serial solve) — exact, sensitive, but tiny. This is exactly the scope's recommended structure.
|
||||
|
||||
---
|
||||
|
||||
## 3. Staged-operand sharing graph (load amortization)
|
||||
|
||||
Five smem/register operands, and which products read them — the fusion that makes the added flops nearly free:
|
||||
|
||||
- **`Kc` (C×dk)** — the most-shared buffer. M-major-over-t LHS: P1, P3. K-major-over-i B (`Kcᵀ`): P1, P2. Transposed (i-major, contract t): P6. ⇒ stage once per chunk, feeds 1,2,3,6.
|
||||
- **`Qc` (C×dk)** — LHS for P2 and P4.
|
||||
- **`S0` B-fragments (dk×dv, register-resident state)** — P3 and P4. **Stage once as B, run KS then QS** (heaviest operand, amortized 2×).
|
||||
- **`Amat` (C×C)** — P1 writes `A` → P7 reads `A` → P2 overwrites with `P` → P5 reads `P`. One buffer, lifecycle-reused (as 0031 already does).
|
||||
- **`Ud` (C×dv)** — P7 writes `U` → P5 reads `U` (B, contract t) → P6 reads `U` scaled to `DU` (B, contract t). **P5 and P6 share the identical `U` B-layout** (both contract the chunk dim) → fully shared B-fragments.
|
||||
|
||||
Three concrete fusions worth coding as fused passes:
|
||||
1. **P1+P2** share `Kcᵀ` B (PoC already does this).
|
||||
2. **P3+P4** share `S0` B (stage the 128×128 state-as-B once).
|
||||
3. **P5+P6** share `U` as B (both K=C contractions); compute `P·U` and `Kcᵀ·DU` from one `U` staging, P6 accumulating straight into the persistent state fragments.
|
||||
|
||||
---
|
||||
|
||||
## 4. tf32-safe vs 3xtf32 ladder - summary + recommended A/B order
|
||||
|
||||
**Plain-tf32-safe (well-conditioned, bounded, intra-chunk; bf16 `m16n8k16` is even an option if more throughput is needed):**
|
||||
- P1 `KK`, P2 `QK` (PoC-proven), P5 `P·U` (P bounded/f32-pre-masked), P7 off-diagonal coupling.
|
||||
|
||||
**3xtf32 / f32 ladder (state-boundary, cross-chunk carry, error compounds):**
|
||||
- P6 `Kᵀ(D·U)` — keep at 3xtf32 longest (compounds over every chunk).
|
||||
- P3 `KS` — feeds the solve; 3xtf32 by default.
|
||||
- P4 `QS` — 3xtf32 by default but γ_t-attenuated → **first to demote** to plain tf32 in the precision A/B.
|
||||
- P7 `16×16` diagonal block inverse — stays **f32/in-register** (not a tensor-core op).
|
||||
|
||||
**Recommended precision A/B ladder (drives the KL-gate from `PAGED_BITEXACT_NOTE.md`):** start P3/P4/P6 at 3xtf32 and P1/P2/P5/P7-offdiag at plain tf32. If the KL-gate has margin, demote in order **P4 → P3**, holding **P6 at 3xtf32**. If even all-3xtf32 misses the KL-gate, the residual is the `16×16` diagonal solve precision, not the mma — that already stays f32.
|
||||
|
||||
---
|
||||
|
||||
## 5. Two honest implementation gotchas (not in the scope doc, surface in the mapping)
|
||||
|
||||
1. **Accumulator→B relayout of the state at each chunk boundary.** The register-resident state lives as P6's **D/accumulator** fragments (`tile<16,8>`), but P3/P4 need it as a **B operand** (`tile<8,8>`, K-major over `i`). These fragment layouts differ, so at chunk entry the state must be re-laid-out accumulator→B. Cheapest correct path: bounce the 128×128 state through a transient smem tile (write D fragments, `ldmatrix` back as B fragments) once per chunk — `n_tokens/C` times total, negligible vs the `C`x state-BW saved, but it means the "freed 64KB" needs a *transient* state-shaped smem tile at boundaries (not resident across the intra-chunk work). With dv-slabbing it's per-slab (`dk×dv_tile`).
|
||||
|
||||
2. **`Kc` needs two fragment views.** P1/P2/P3 read `Kc` contracting over `i` (dk); P6 reads `Kc` contracting over `t` (transposed, i as M-row). One smem buffer, but P6 must use the `ldmatrix` transpose / J-major `tile` view — budget for the transposed load, don't assume one staging serves both.
|
||||
|
||||
Ordering within a chunk is already correct in 0031 and must be preserved by the tensor-core version: P3,P4 read **pre-update** `S0` → P7 solve → P5 → **P6 overwrites** `S0`→`S_C`. Single accumulator, read-first/write-last, no state double-buffer needed.
|
||||
|
||||
## 2. A-inverse solve (form-T then apply, FLA UT transform)
|
||||
|
||||
I have everything I need: the chunked GDN math (patch 0031), the scope doc's recommendation, the ggml tf32 `mma.sync` API (`mma(tile<16,8,float>&D, tile<16,8,float>&A, tile<8,8,float>&B)` = m16n8k8), and the proven Gram PoC layout (`g=lane>>2, t=lane&3`; tf32 NMSE ~3e-9). Here is the design.
|
||||
|
||||
---
|
||||
|
||||
# A-inverse solve on `mma.sync` tensor cores (C=64, sm_121a) - design
|
||||
|
||||
Notation: `C=64`, head dim `dk=dv=128`, block size `b=16` (= one `m16n8k8` m-tile), `n_b=C/b=4`. `A = I + N`, `N = tril(beta_t·d(t',t)·(k_t·k_t'), -1)` strictly-lower (nilpotent, `N^C=0`); `RHS[t][j] = beta_t(v_t[j] - gamma_t(S0^T k_t)[j])` is `C×dv`; we want `U = A^{-1}·RHS`.
|
||||
|
||||
## 0. Core decision: form `T=A^{-1}` explicitly, then one wide apply (not direct back-subst)
|
||||
|
||||
Two routes were on the table. **Form `T = A^{-1}` in the `C×C` domain (FLA "UT transform"), then `U = T·RHS` as a single tf32 GEMM** - rather than blocked forward-substitution applied directly to the `C×dv` RHS. Reasons, all decisive on this part:
|
||||
|
||||
1. **Confines the only triangular dependency to the cheap `C×C` domain.** The expensive `dv=128`-wide work (`U=T·RHS`) becomes a dependency-free dense GEMM. The serial part is just the tiny `T`-formation. This is the single most important move for "don't serialize the warps."
|
||||
2. **Fewer serial passes vs `dv`.** Inverting the `16×16` diagonal block once = a 16-column solve. Direct-solving against `RHS` re-solves against all `dv=128` columns per block. Form-`T`-once + reuse via mma is far cheaper in serial work.
|
||||
3. **dv-slab reuse (the occupancy lever).** `T` depends only on `K`, not on `dv`. Form once, reuse for every `dv`-slab's `T·RHS_slab` apply. (Improvement over the scope's conservative "recompute per slab": when single-block, `T` lives in 16KB shared and is broadcast; only when dv-slabbing across separate blocks for occupancy do we recompute - which is cheap anyway, ~12% of the apply's mma count.)
|
||||
4. **Isolates the error amplifier.** All recursion (the part that "amplifies error") lives in the small `T`-formation where 3xtf32 is nearly free; the bulk apply is a single well-conditioned GEMM.
|
||||
|
||||
This still **is** the scope's "blocked forward substitution: in-register diagonal solves + mma off-diagonal coupling" - just organized to produce `T` explicitly so the wide apply is dependency-free.
|
||||
|
||||
## 1. Solve algorithm
|
||||
|
||||
Block-partition `A` into a `4×4` lower-triangular grid of `16×16` blocks. `A_ii = I_b + N_ii` (unit-lower-tri, `N_ii` strictly-lower nilpotent); `A_ij` (i>j) full `16×16`. `T=A^{-1}` is block-lower-tri with:
|
||||
|
||||
```
|
||||
T_ii = A_ii^{-1} (diagonal block inverse)
|
||||
T_ij = -A_ii^{-1} · ( Σ_{m=j}^{i-1} A_im · T_mj ) for i > j (block fwd subst)
|
||||
```
|
||||
|
||||
Then `U = T·RHS`, with `U_i = Σ_{j≤i} T_ij·RHS_j`.
|
||||
|
||||
**Phase D - diagonal inverses (4 blocks, fully parallel).** Each `A_ii` is `16×16` unit-lower-tri. Invert **exactly in f32** via shared-memory column-parallel forward substitution: stage `A_ii` to shared; thread `c` (c=0..15) solves `A_ii x = e_c` (`x_c=1`, `x_r = -Σ_{m=c}^{r-1} A_ii[r][m]·x_m`), writes column `c` of `T_ii`. 16 columns in parallel, ≤16 serial MACs each, all 4 blocks on 4 warps simultaneously. **No tensor cores here, and no reduced precision** - this is where the strongest coupling lives (see §4).
|
||||
|
||||
**Phase O - off-diagonal, mma.** For each i>j: accumulate `P_ij = Σ_m A_im·T_mj` (δ block-products), then `T_ij = -T_ii·P_ij`. All on `mma.sync` (`16×16×16` = `2 n-tiles × 2 k-steps` = 4 m16n8k8 per block-product).
|
||||
|
||||
**Apply.** `U = T·RHS`: warp `w` owns output rows `16w..16w+15`, sweeps all `dv=128` (16 n-tiles) × `C=64` (8 k-steps) = 128 m16n8k8/warp. This is the bulk and is embarrassingly parallel.
|
||||
|
||||
`A`, `P` (the QK term), `RHS`, and `T` are all assembled from tf32 Gram mma's (`KK`,`QK`,`KS`,`QS` - the PoC-proven step-1/2 plus step-3/4) with **all decay/`gamma`/`beta` applied in f32 outside the mma** (preserves bounded de-gating).
|
||||
|
||||
## 2. Tile schedule - keeping the triangular dependency off the warps
|
||||
|
||||
Block = 128 threads = 4 warps; **"warp == 16-row m-tile" throughout** (same mapping as the PoC's C=64 kernel, `rowbase = warp*16`, `g=lane>>2`, `t=lane&3`). Three layered mechanisms keep the warps busy despite the triangular dependency:
|
||||
|
||||
**(a) Wavefront (anti-diagonal) parallelism in `T`-formation.** The 6 off-diagonal blocks have a critical path of only `n_b-1=3`, not 6. Group by distance `δ=i-j`:
|
||||
|
||||
| Wave | Blocks (δ) | count | depends on | mapped to |
|
||||
|---|---|---|---|---|
|
||||
| D | (0,0)(1,1)(2,2)(3,3) | 4 | - | 4 warps ‖ |
|
||||
| 1 | (1,0)(2,1)(3,2) | 3 | D | 3 warps ‖ |
|
||||
| 2 | (2,0)(3,1) | 2 | D,1 | 2 warps ‖ |
|
||||
| 3 | (3,0) | 1 | D,1,2 | 1 warp |
|
||||
|
||||
Within each wave all blocks are independent → one block per warp, no intra-wave serialization. Critical path = 4 dependency levels. Total `T`-formation mma: ~10 accumulation block-products + 6 inverse-applies = ~16 block-products × 4 = **~64 m16n8k8**, vs the apply's **512** (128/warp × 4) - so `T`-formation is ~12% of apply width and carries the only dependency.
|
||||
|
||||
**(b) Confinement.** Because we form `T` then apply, the dependency-laden work is the ~64-mma `C×C` formation; the 512-mma `dv`-wide apply has zero triangular dependency. The serial chain never touches the throughput-critical width.
|
||||
|
||||
**(c) Latency hiding via RHS overlap.** `T` depends only on `K` (→ `A` ← KK Gram). `RHS` depends on `V` and `S0^T k` (KS Gram, `dv`-wide, the expensive RHS term) and is **independent of the solve**. Schedule the wavefront `T`-formation (cheap, short critical path) concurrently with the `dv`-wide KS/QS Grams that build `RHS` and the `O` cross-term. The Phase-D shared scalar inverse (~16 shared round-trips × 4 warps) hides entirely under the KS Gram (thousands of cycles). By the time `T` is ready, `RHS` is staged and the apply fires immediately.
|
||||
|
||||
**Shared/register budget (C=64, state register-resident per the scope):**
|
||||
|
||||
| Buffer | bytes | note |
|
||||
|---|---|---|
|
||||
| `Kc`,`Qc` (bf16/tf32-staged) | 16KB+16KB | Gram operands |
|
||||
| `A`→`T` scratch (`C×C` f32, in place) | 16KB | `A` consumed into `T`; reuses scope's A/P slot |
|
||||
| `RHS`/`U` (`C×dv`) | 16-32KB | bf16 for the P·U and KᵀU mma's |
|
||||
| diag-inverse scratch | ~1KB | `16×16` per warp, transient |
|
||||
| gates `cs/gam/beta` | <1KB | f32 |
|
||||
| state `S` | 0 (registers) | frees the 64KB that forced 0031's C=16 |
|
||||
|
||||
Total ~65-80KB, under the 99KB opt-in - the solve adds **no** net shared pressure (T overwrites A; diag scratch is transient). Per-thread diag-inverse needs ~16 regs (one column of `x`), released before the apply - does not compound the already-heavy state-accumulator register budget.
|
||||
|
||||
## 3. Precision risk assessment
|
||||
|
||||
**Error model.** `‖ΔU‖/‖U‖ ≲ κ(A)·(‖ΔA‖/‖A‖ + ‖ΔRHS‖/‖RHS‖) + ‖Δ_apply‖/‖U‖`. The inverse is the amplifier; `κ(A)` is data-dependent. For DeltaNet, keys are L2-normalized so `|k_t·k_t'|≤1` ⇒ `|N[t][t']|≤beta_t≤1`; in the decaying regime `‖N‖<1` and `κ` is modest, but in the weak-decay/aligned-keys corner `κ` grows and the `δ=3` column path (`T_30`) compounds 3 multiplies. tf32 input rounding is ~`2⁻¹¹`≈`5e-4` relative (f32-accumulate; PoC measured Gram NMSE ~`3e-9`). 3xtf32 (3-limb split, the CUTLASS fp32-emulation trick) buys ~f32 (~`1e-7`) at ~3× that step's mma cost.
|
||||
|
||||
**Where the strong coupling actually sits (the key structural fact):** the *inverse* `T_ii` is computed f32-exact, **but the dominant near-diagonal mixing is applied in the tf32 apply GEMM** (`U_i ⊃ T_ii·RHS_i`), and block-boundary adjacent pairs (e.g. tokens 15↔16) live in the `δ=1` off-diagonal `T_10`. So "f32 protects the strong coupling" is only true for the inverse *computation*; its *application* is tf32 unless promoted. This drives the ladder.
|
||||
|
||||
**Precision config + 3xtf32 ladder (mandatory vs optional):**
|
||||
|
||||
| Step | Default | Mandatory? | 3xtf32 cost |
|
||||
|---|---|---|---|
|
||||
| Diagonal inverse `T_ii` | **f32 (shared scalar)** | **Mandatory-and-free** (it's already f32) | n/a |
|
||||
| Off-diag coupling `A_im·T_mj`, `T_ii·P_ij` | **3xtf32 (default-on)** | Effectively mandatory; ~3× of ~64 tiny mma = **negligible** | free insurance |
|
||||
| KK/QK Gram → A,P | tf32 | optional (rung 1) | 3× of C×C Grams (cheap) |
|
||||
| Apply `U=T·RHS` | tf32 | optional (rungs 2-4) | up to 3× the bulk |
|
||||
| KS/QS Gram → RHS, O | tf32 | optional (rung 5) | vLLM keeps these bf16 (L4-rejected precedent) |
|
||||
|
||||
Decays/`gamma`/`beta` **always f32, outside the mma** - invariant, not a rung.
|
||||
|
||||
**Ladder ordering if the default config misses the KL-gate (cheapest → most expensive):**
|
||||
1. KK Gram (feeds `A`) → 3xtf32 [cheap, C×C].
|
||||
2. Apply **block-diagonal terms only** `T_ii·RHS_i` → 3xtf32 [≈+0.8× apply; protects within-window strong coupling - mixed-precision-by-distance].
|
||||
3. + apply `δ=1` off-diagonal terms → 3xtf32 [covers block-boundary adjacent pairs].
|
||||
4. Full apply → 3xtf32 [≈+2× apply; expensive escape hatch].
|
||||
5. KS/QS Gram → 3xtf32.
|
||||
6. Fall back to direct blocked back-substitution against RHS in 3xtf32 (the alternative route, slightly more accurate than form-`T`-then-multiply at the cost of the parallelism), else keep 0031's serial path.
|
||||
|
||||
**Adversarial `g∈[-20,-1e-4]`:** strong decay ⇒ `d=exp(big-negative)→0` ⇒ off-diagonal `N→0` ⇒ `A≈I`, `T≈I`, apply≈identity, tf32 error vanishes; bounded de-gating (f32) guarantees underflow-to-zero, never inf. Weak decay (`g→0`) ⇒ `d≈1`, `A` well-conditioned, tf32's 8-bit exponent (vs f16's 5) holds the `gamma` dynamic range. The dangerous middle is the only KL-empirical risk - re-run this op case explicitly per the scope.
|
||||
|
||||
**KL impact / gating.** Same gate as the backend's new-FP-path precedents: NMSE is expected to *fail* at reduced precision (this is a new path on a new path) - **the binding gate is KL** (`KLD(tc‖f16) ≤ KLD(seq‖f16)` + PPL band) plus greedy-md5 stability (md5 will not match 0031's serial path - per-path, validated benign). Expectation: the **default config (f32 diagonal + 3xtf32 off-diagonal-coupling + tf32 everything-else)** clears the KL-gate, because (i) the dominant apply matches the PoC Gram's `~3e-9`/tf32-input grade and (ii) the recursion-amplified `C×C` work is f32-grade for free. The expensive apply-3xtf32 rungs are reserved escapes. Worst case all-3xtf32 ≈ 3× the mma cost - still an order of magnitude under 0031's serial-f32 reductions and still net-positive given the `~C×` state-BW cut.
|
||||
|
||||
## 4. Integration + validation
|
||||
|
||||
- Build on `ggml/src/ggml-cuda/mma.cuh`: the tf32 path is `mma(tile<16,8,float>&D, tile<16,8,float>&A, tile<8,8,float>&B)` → `mma.sync.aligned.m16n8k8.row.col.f32.tf32.tf32.f32` (line ~1089), gated by `AMPERE_MMA_AVAILABLE` (sm_121-correct). tf32 operands stage to shared and load via `load_generic` (or the PoC's `cvt.rna.tf32.f32` register packing); `ldmatrix` is `.b16`-only so it is **not** usable for tf32 fragments - use `load_generic`. 3xtf32 = 3 limb-split tf32 mma's + f32 recombine, reusing the same tile type.
|
||||
- Slots into patch 0031's `gated_delta_net_chunked_cuda`: replace the serial `KK`/`QK` Grams (lines 183-198, 222-235) with the PoC tiles; replace the serial forward-subst (lines 211-218) with this Phase-D + Phase-O `T`-formation; replace the serial `O`/`S_C` reductions (steps 5/6) with the apply + `KᵀU` mma. `C=16`→`C=64`, state register-resident.
|
||||
- Validation step before the full kernel: extend `~/scratch_tc_gdn_poc/gdn_gram_bench.cu` (DGX) to microbench Phase-D+O `T`-formation and the `T·RHS` apply in isolation, with the f64 host oracle measuring `κ(A)` and the tf32-vs-3xtf32 NMSE per rung - mirrors the Gram PoC that proved 6.7×→9.3×. This tells you, cheaply, whether the default precision config will clear the KL-gate before committing to the multi-week kernel (Phase 2 of the scope's plan).
|
||||
|
||||
**Net:** the solve adds zero shared pressure, ~12% mma overhead over the apply, a 4-level (not 64-step) critical path fully hidden under the RHS Grams, an f32-exact inverse, and a cheap-by-default 3xtf32 placement aimed exactly at the recursion-amplified term - so the A-inverse is **not** the bottleneck and **not** the precision wall; the residual risk is the data-dependent `κ(A)` in the weak-decay corner, arbitrated by the KL-gate with the apply-3xtf32 rungs as the escape.
|
||||
|
||||
Files referenced (absolute): `/home/mudler/_git/LocalAI/.claude/worktrees/feat+paged-attention/backend/cpp/llama-cpp-localai-paged/docs/TENSORCORE_GDN_SCOPE.md`; `/home/mudler/_git/LocalAI/.claude/worktrees/feat+paged-attention/backend/cpp/llama-cpp-localai-paged/patches/paged/0031-paged-chunked-gdn-prefill-scan-kernel.patch`; ggml tf32 mma at `ggml/src/ggml-cuda/mma.cuh` (line ~1089); DGX PoC `~/scratch_tc_gdn_poc/gdn_gram_bench.cu`.
|
||||
|
||||
## 3. Register-resident state + C=64 budget
|
||||
|
||||
I have everything I need: the scope doc, the full 0031 kernel (smem layout, the 7 product families, the chunk-loop sequencing), and the sibling FP4-MMA GEMM kernel's concrete patterns (PAD=4 u32 rows, STAGES=2 cp.async, WM=4/WN=2 8-warp grid, ldmatrix, register-resident accumulators). Here is the concrete data-layout design.
|
||||
|
||||
---
|
||||
|
||||
# Register-resident state + C=64 design (tensor-core chunked GDN, sm_121a / mma.sync)
|
||||
|
||||
## 0. Block/warp configuration (the frame everything hangs on)
|
||||
|
||||
0031 used `block_dims(128,1,1)` = 128 threads, "thread j owns v-column j" - a **column-parallel scalar** model. The tensor-core kernel must abandon that and adopt a **warp-tiled** model (same as the sibling GEMM kernel):
|
||||
|
||||
- **256 threads = 8 warps**, arranged as a **WARPS_M x WARPS_N = 4 x 2** warp grid (the GEMM kernel's proven `WM=4, WN=2`).
|
||||
- Threads no longer own columns; warp `(wm,wn)` owns a rectangular sub-tile of each matrix and drives `mma.sync` on it.
|
||||
- Precision: **tf32 m16n8k8** for the S-coupled / decay-coupled products, **bf16 m16n8k16** allowed only for the well-conditioned intra-chunk Gram terms (KK, QK). f32 accumulate throughout. Decays/`gamma`/`beta` stay f32, applied outside the mma (preserve bounded de-gating).
|
||||
|
||||
This 4x2 warp grid is the denominator for every ownership calc below.
|
||||
|
||||
---
|
||||
|
||||
## 1. The one hard problem: S is an *accumulator* for step 6 but an *operand* for steps 3/4
|
||||
|
||||
This is the crux the scope hand-waves ("read S as the stationary operand; step 6 accumulates into it"). The register fragment layouts are **not** interchangeable:
|
||||
|
||||
| Use | Role | mma shape | S indexing | Fragment layout |
|
||||
|---|---|---|---|---|
|
||||
| Step 6 `S += Kᵀ(D·U)` | **accumulator (D/C)** | m=dk, n=dv, k=C | `S[i][j]`, m=i, n=j | `tile<16,8,float>` acc grid |
|
||||
| Step 3 `KS = K·S` | **B operand** | m=C, n=dv, k=dk | `S[i][j]`, k=i, n=j | `tile<8,8,float>` B frag |
|
||||
| Step 4 `QS = Q·S` | **B operand** | m=C, n=dv, k=dk | same as step 3 | `tile<8,8,float>` B frag |
|
||||
|
||||
An accumulator fragment's thread→element map differs from a B-operand fragment's, so **you cannot feed the persistent S registers directly into the step-3/4 mma.** A bridge is mandatory. The design decision:
|
||||
|
||||
> **S lives register-resident in the step-6 ACCUMULATOR layout** (it is written every chunk; that is the hot path). Steps 3/4 reach it via a **once-per-chunk restage to a small smem tile, re-read with `ldmatrix`** as B-operand fragments.
|
||||
|
||||
The restage cost is paid `n_tokens/C` times (not per token) - it is *inside* the BW saving the whole lever buys. And critically, the restage smem **time-multiplexes onto the Uc/Amat region**: at chunk entry (when KS/QS are needed) U and A for this chunk are not yet computed, so their buffers are free to hold the S restage. **Net additional persistent smem for the state: 0KB** - the scope's "0KB shared state" holds, with this scheduling caveat made explicit.
|
||||
|
||||
KS and QS read the **same** pre-update S0, so restage once → do both → then overwrite with U.
|
||||
|
||||
---
|
||||
|
||||
## 2. Register allocation map (per thread, 256-thread block)
|
||||
|
||||
State `S` is `dk x dv = 128 x 128` f32 = 16384 elems. Distributed over 256 threads = **64 f32/thread** at full dv.
|
||||
|
||||
| Register class | Lifetime | Full dv=128 | dv-slab=64 | dv-slab=32 | Layout / ownership |
|
||||
|---|---|---|---|---|---|
|
||||
| **Persistent S accumulator** | whole chunk loop | **64 regs** | **32 regs** | **16 regs** | warp `(wm,wn)` owns dk-rows `[wm·32, +32)` x dv-cols `[wn·(dv/2), +dv/2)`; = 2 m-tiles x (dv/2/8) n-tiles of `tile<16,8,float>`, 4 f32 each |
|
||||
| Transient A-operand frag | per product | 4 regs/tile | same | same | `tile<16,8,float>` (tf32 packs k8) reused across KK/QK/KS/QS/O/Supd |
|
||||
| Transient B-operand frag | per product | 2 regs/tile | same | same | `tile<8,8,float>` |
|
||||
| Transient product accumulator (KK/QK/KS/QS/O) | per product, then spilled to smem | ≤8 tiles·4 = 32 regs | ≤16 | ≤8 | these outputs go to smem; acc is transient, reused |
|
||||
| A⁻¹ diagonal-block solve (16x16, in-registers) | step 7 only | ~8-12 regs | same | same | one `b=16` unit-lower-tri block per warp-row, scalar Neumann/`<b-1` terms |
|
||||
| loop/index/gate scalars | always | ~12 regs | same | same | c0, Cc, cs/gam/beta locals |
|
||||
|
||||
**Per-thread totals (256 threads):**
|
||||
- Full dv: 64 (S) + ~50 (transients, non-overlapping with S) + ~12 ≈ **~130 regs/thread** → fits **1 block/SM** (256 regs/thread budget at 65536/SM÷256). 2 blocks/SM (128 regs/thread cap) would spill - hence dv-slab for occupancy.
|
||||
- dv-slab 64: 32 (S) + ~50 + 12 ≈ **~94 regs/thread** → fits **2 blocks/SM** (128-reg cap). ✓
|
||||
- dv-slab 32: 16 (S) → ~78 regs/thread → headroom; grid x4.
|
||||
|
||||
Persistent-state register pressure is the occupancy gate; everything else is transient and reused across the 7 products.
|
||||
|
||||
---
|
||||
|
||||
## 3. Shared-memory allocation map (PAD-padded, conflict-free)
|
||||
|
||||
Apply the GEMM kernel's lesson verbatim: **PAD = 4 (in the row's element width)** so a 128-wide row (a multiple of the 32 banks → 8-way conflict for the 8-row `ldmatrix`) becomes stride `132`, and the 8 rows of an `ldmatrix.m8n8` land in 8 distinct banks: `(r·132 + c) mod 32 = (4r + c) mod 32`, distinct for `r=0..7`. ✓
|
||||
|
||||
| Buffer | Logical shape | Row stride (padded) | Element | Notes |
|
||||
|---|---|---|---|---|
|
||||
| `Kc` (chunk K) | `[C][dk]` | `dk + 8` bf16 (= +4 u32) | bf16 | A-operand for KK/QK; A/Bᵀ for KS; transposed-A for Supd. bf16 default |
|
||||
| `Qc` (chunk Q) | `[C][dk]` | `dk + 8` bf16 | bf16 | A-operand for QK/QS/O |
|
||||
| `Uc` (solved U) | `[dv][C]` | `C + 4` f32 | f32 | f32 for the triangular solve accuracy; B-operand (down-cast tf32) for O & Supd |
|
||||
| `Amat` (A then P) | `[C][C]` | `C + 4` f32 | f32 | KK→A→solve, then reused for QK→P; decays applied in f32 here |
|
||||
| `gates` cs/gam/beta | `[3·C]` | (1-D, no pad) | f32 | prefix-sum + `expf`, f32 always |
|
||||
| **S-restage tile** | `[dk][dv_strip]` | `dv_strip + 4` f32 | f32 | **overlays Uc∪Amat** at chunk entry; not additive at peak |
|
||||
| `cp.async` stage dup | (STAGES=2 on Kc/Qc) | as Kc/Qc | bf16 | Phase-3 latency hiding only |
|
||||
|
||||
PAD widths: f32 tiles +4 elems; bf16 tiles +8 elems (= +4 u32, identical bank offset as the GEMM kernel). `Uc` and `Amat` are padded on the C dimension (their `ldmatrix` access dimension).
|
||||
|
||||
---
|
||||
|
||||
## 4. C=64 shared budget table (under the 99KB opt-in)
|
||||
|
||||
Byte math with PAD included (`KB = bytes/1024`):
|
||||
|
||||
| Buffer | CONFIG A — **default**: C=64, dv=128, K/Q **bf16** | CONFIG B: C=64, dv=128, K/Q **tf32** | CONFIG C — **2 blk/SM**: C=32, dv-slab=64, K/Q bf16 |
|
||||
|---|---|---|---|
|
||||
| `Kc` | 64·136·2 = **17.0KB** | 64·132·4 = 33.0KB | 32·136·2 = 8.5KB |
|
||||
| `Qc` | **17.0KB** | 33.0KB | 8.5KB |
|
||||
| `Uc` (f32) | 128·68·4 = **34.0KB** | 34.0KB | 64·36·4 = 9.0KB |
|
||||
| `Amat` (f32) | 64·68·4 = **17.0KB** | 17.0KB | 32·36·4 = 4.5KB |
|
||||
| gates | **0.75KB** | 0.75KB | 0.4KB |
|
||||
| S-restage | overlay (0 net) | overlay (0 net) | overlay (0 net) |
|
||||
| **Per-block total** | **≈ 85.8KB** ✅ < 99 | **≈ 117.8KB** ❌ | **≈ 30.9KB** |
|
||||
| Blocks/SM (≈100KB/SM) | **1** | n/a | **2** (61.8KB) ✅ |
|
||||
|
||||
Read-out:
|
||||
- **CONFIG A is the recommended default**: C=64 (4x the 0031 chunk), full dv, fits at ~86KB with margin, 1 block/SM. Peak is the O/Supd phase (all of Kc+Qc+Uc+Amat live).
|
||||
- **CONFIG B (tf32 K/Q) is budget-hostile** (117KB) - tf32 K/Q tiles don't shrink with dv-slab (they're `C x dk`), so even dv-slab=64 lands ~101KB. **Conclusion: stage K/Q as bf16; reserve tf32/3xtf32 for the S-coupled and decay-coupled terms** (which arrive via the small streamed S-restage and the f32 gate scaling), exactly per the scope's "bf16 only for well-conditioned Gram terms."
|
||||
- **CONFIG C is the 2-block/SM lever**: C=32 + dv-slab=64 → 31KB/block, two resident blocks under the ~100KB/SM total, and the grid grows to `H x n_seqs x 2`.
|
||||
|
||||
---
|
||||
|
||||
## 5. dv-slab strategy (the 2nd block/SM + grid-starvation fix)
|
||||
|
||||
Split the `dv=128` value dimension into `n_slabs` blocks; each block computes a `dv_tile`-wide vertical strip of O and of the state.
|
||||
|
||||
- **Grid**: `dim3(H, n_seqs, n_slabs)` (was `(H, n_seqs)`). `n_slabs ∈ {1,2,4}` for `dv_tile ∈ {128,64,32}`. This **multiplies the grid by `n_slabs`**, directly attacking 0031's low-`n_seqs` grid starvation.
|
||||
- **What is dv-independent and must be recomputed per slab**: `A` (KK Gram), the `A⁻¹` solve, the gate prefix - all depend only on K and the gates, *not* dv. Each slab recomputes them. Cheap once they are on tensor cores (the whole point); this is the FLA per-slab pattern.
|
||||
- **What is dv-sliced**: the S accumulator (`128 x dv_tile`), `Uc` (`dv_tile x C`), KS/QS/O outputs, step-6 update. Halving/quartering dv halves/quarters both the **S register footprint** (64→32→16 regs/thread, §2) and the dv-scaled smem (`Uc`, restage).
|
||||
- **Restage budget bonus**: at `dv_tile=64` the per-block S is `128 x 64` = 32KB, so the once-per-chunk restage fits the Uc∪Amat overlay window in a single pass (no strip loop). At full dv=128 the restage is done as **2 dv-strips of 32KB** reusing the same overlay (or 16 k-strips of 8x128 if registers are tighter than smem).
|
||||
|
||||
`b`-block forward substitution (step 7) is independent of dv too, so the in-register `16x16` diagonal solves are computed once and the off-diagonal mma coupling `Uᵢ -= Aᵢⱼ Uⱼ` runs per slab as a `(16x16)·(16 x dv_tile)` mma.
|
||||
|
||||
---
|
||||
|
||||
## 6. Bank-conflict-free layout - the GEMM PAD lesson applied
|
||||
|
||||
Concretely, per the sibling kernel's `ARS = KBLK·SAW + PAD` with `PAD=4` (which gave +19%):
|
||||
|
||||
- Every smem matrix read by `ldmatrix` (or its tf32 equivalent in `ggml/src/ggml-cuda/mma.cuh`) is stored with **row stride = logical_width + PAD**, PAD chosen so `stride mod 32 ≠ 0`: f32 width-128 → 132 (`132 mod 32 = 4`); bf16 width-128 (packed 64 u32) → 68 u32 (`68 mod 32 = 4`).
|
||||
- This guarantees the 8 rows an `m8n8` `ldmatrix` touches map to 8 distinct banks for any fixed column → no replays on the operand loads, which are the kernel's inner-loop smem traffic.
|
||||
- `cp.async` (CONFIG, Phase 3): `STAGES=2` double-buffer on `Kc`/`Qc` only (the GEMM kernel found multistage saturates BW past depth 2). 16B `cp.async.cg` copies into the padded rows; `commit_group`/`wait_group` Ampere-style (no TMA on sm_121). The pad keeps the staged writes coalesced and the mma reads conflict-free simultaneously.
|
||||
|
||||
---
|
||||
|
||||
## 7. Summary of the allocation decisions
|
||||
|
||||
| Decision | Value |
|
||||
|---|---|
|
||||
| Threads / warp grid | 256 / 4x2 (WM=4, WN=2) |
|
||||
| **State residency** | register-resident in **step-6 accumulator layout** (`tile<16,8,float>` grid), 64/32/16 f32-regs/thread at dv 128/64/32 |
|
||||
| **Accumulator↔operand bridge** | once-per-chunk `ldmatrix` restage of S to a small smem tile that **overlays Uc∪Amat** (0 net persistent smem); KS+QS share one restage |
|
||||
| K/Q precision | **bf16** staged (tf32 K/Q breaks the 99KB budget); tf32/f32 reserved for S-coupled + decay-coupled terms |
|
||||
| Uc / Amat | f32, padded on C (+4) |
|
||||
| **PAD** | +4 f32 (+8 bf16 = +4 u32) row stride → `ldmatrix` 8-row conflict-free (GEMM-proven) |
|
||||
| **C=64 default budget** | **≈86KB**, 1 block/SM (CONFIG A) ✅ |
|
||||
| 2 blocks/SM | C=32 + dv-slab=64 → ≈31KB/block, grid x2 (CONFIG C) |
|
||||
| dv-slab | grid `(H, n_seqs, n_slabs)`; A/A⁻¹/gates recomputed per slab; S/Uc/O dv-sliced |
|
||||
| cp.async | STAGES=2 on Kc/Qc (Phase 3 only) |
|
||||
|
||||
One honest caveat surfaced beyond the scope doc: the scope's "~64-80KB / 0KB shared state" budget only holds with **bf16 K/Q staging** and the **overlay restage**; tf32 K/Q (CONFIG B) does not fit even with dv-slab, which is why bf16-Gram + tf32-only-for-S/decay is the forced (and scope-consistent) precision split.
|
||||
|
||||
Files referenced: `/home/mudler/_git/LocalAI/.claude/worktrees/feat+paged-attention/backend/cpp/llama-cpp-localai-paged/docs/TENSORCORE_GDN_SCOPE.md`, `/home/mudler/_git/LocalAI/.claude/worktrees/feat+paged-attention/backend/cpp/llama-cpp-localai-paged/patches/paged/0031-paged-chunked-gdn-prefill-scan-kernel.patch`, and the sibling GEMM patterns in `/home/mudler/_git/LocalAI/.claude/worktrees/feat+paged-attention/backend/cpp/llama-cpp-localai-paged/patches/paged/0034-feat-paged-native-NVFP4-W4A4-FP4-MMA-large-M-prefill.patch`.
|
||||
|
||||
## 4. Occupancy + launch config
|
||||
|
||||
I have everything I need. Key facts confirmed: tf32 tile is `mma.sync.aligned.m16n8k8.f32.tf32.tf32.f32` (mma.cuh:1089, accumulator `tile<16,8,float>` = 4 f32/thread/tile); grid.x = `H` = `n_v_heads` = `ssm_dt_rank` (llama-model.cpp:504; Qwen3-Next family = 32 GDN value heads); the sequential kernel already uses a 3D grid `(H, n_seqs, ceil(S_v/num_warps))` (gated_delta_net.cu:184) — the chunked 0031 collapsed that z-axis to 1, which is exactly what starves the grid.
|
||||
|
||||
Here is the occupancy + launch design.
|
||||
|
||||
---
|
||||
|
||||
# Occupancy + launch design — tensor-core chunked GDN prefill (sm_121a)
|
||||
|
||||
## 0. The two independent caps 0031 hit (must relieve BOTH for ≥2 blocks/SM)
|
||||
|
||||
0031's -22% is not one wall, it's two stacked walls, and they are relieved by *different* levers:
|
||||
|
||||
| Cap | 0031 value | Binding resource | Lever |
|
||||
|---|---|---|---|
|
||||
| **Shared-memory cap** | 89 KB (64 KB all-shared state) | 100 KB/SM, 99 KB dyn opt-in | state→registers **+ smaller C** |
|
||||
| **Register cap** | n/a (was scalar) | 65536 regs/SM | **dv-slab** the register-resident state |
|
||||
| **Grid cap** | `(H, n_seqs, 1)` = 32·n_seqs blocks | 48 SMs | **dv-slab multiplies grid** by n_slabs |
|
||||
|
||||
sm_120/121-class per-SM limits used throughout: **1536 threads/SM, 65536 32-bit regs/SM, 100 KB shared/SM (99 KB dynamic opt-in), 255 regs/thread, ≤24-32 blocks/SM (hw, never the binding limit here).** The binding limits are **shared and registers.**
|
||||
|
||||
Critical correction to the scope-doc budget table: it assumes **bf16** K/Q staging (2 B). The precision default is **tf32**, which is a 32-bit container in shared — tf32 K/Q would *double* Kc/Qc and blow C=64 past 99 KB. So the occupancy config **stages K/Q as bf16** (the well-conditioned KK/QK Gram products per the scope's "bf16 only for Gram terms"), keeps gates/decays/beta/the solve in f32. This is a real precision↔occupancy coupling, flagged in §5.
|
||||
|
||||
## 1. Grid mapping — three parallel axes, the chunk axis is serial
|
||||
|
||||
The inter-chunk recurrence carries state `S` across chunks, so **the chunk axis cannot be a grid axis** (it's the sequential dependency — that's the whole algorithm). The only legitimate grid axes that don't break the recurrence are:
|
||||
|
||||
```
|
||||
dim3 grid(H, n_seqs, n_slabs); // H = n_v_heads = 32 (ssm_dt_rank)
|
||||
// n_slabs = dv / dv_tile (the new lever)
|
||||
```
|
||||
|
||||
- `blockIdx.x = head` (0..31), `blockIdx.y = seq`, `blockIdx.z = dv-slab`.
|
||||
- A block owns v-columns `[z·dv_tile, (z+1)·dv_tile)`, walks the chunk loop serially, and keeps **only its `dk × dv_tile` state slab** register-resident.
|
||||
- This reuses the **same 3D grid shape the sequential kernel already has** (gated_delta_net.cu:184 uses z for S_v-splitting); the chunked kernel repurposes z from S_v-split to dv-slab. The dispatcher change is minimal.
|
||||
|
||||
**Saturation math (the core of the task).** Target ≥2 blocks/SM on 48 SMs ⇒ **≥96 concurrent blocks**. With H=32:
|
||||
|
||||
| n_seqs | dv_tile=128 (n_slabs=1) | dv_tile=64 (2) | dv_tile=32 (4) |
|
||||
|---|---|---|---|
|
||||
| 1 | 32 (starved, 0031) | 64 (48/48 SMs busy, 67% warp-occ) | **128 (100%)** |
|
||||
| 2 | 64 | **128 (100%)** | 256 |
|
||||
| 4 | 128 | 256 | 512 |
|
||||
|
||||
So **dv-slabbing is simultaneously the register-relief lever and the grid-multiplier** — it's the single most important move. Rejected grid alternatives: split-K over dk (needs cross-block atomic reduction + fights the state carry); batching heads/seqs per block (reduces grid, wrong direction).
|
||||
|
||||
## 2. Block dim / warp count — 8 warps / 256 threads
|
||||
|
||||
```
|
||||
constexpr int WARPS = 8;
|
||||
dim3 block(32 * WARPS, 1, 1); // 256 threads
|
||||
```
|
||||
|
||||
Why 8 warps:
|
||||
- **Clean mma tile partition at C=32:** KK/QK output is `C×C = 32×32` = (32/16)·(32/8) = **8 m16n8 tiles → exactly 1 tile/warp**, dk=128 = 16 k8-steps. Steps 3/4 (KS/QS) and 5 (P·U) → 2 tiles/warp. Step 6 state update `dk×dv_tile`=128×64 → 64 tiles → **8 tiles/warp** (these are the persistent register-resident accumulators).
|
||||
- **Register dilution:** the register-resident state accumulator is spread across all 256 threads (see §3) — more warps = fewer state-regs/thread.
|
||||
- **Threads are not the cap:** 256 threads ⇒ up to 6 blocks/SM by the 1536 thread limit, so registers/shared decide.
|
||||
|
||||
Fallback if register-capped (§5): **12 warps (384 threads)** dilutes the state accumulator further (dv_tile=64: 32→21 state-regs/thread) at the cost of thinner per-warp tiles and ≤4 blocks/SM by threads.
|
||||
|
||||
## 3. Register-resident state ↔ dv-slab ↔ occupancy interaction
|
||||
|
||||
The state slab is held as **tf32 mma accumulator fragments** (`tile<16,8,float>`, 4 f32/thread/tile) persisting across the chunk loop. Per-thread state-register cost = `dk·dv_tile / 256`:
|
||||
|
||||
| dv_tile | state f32/block | state regs/thread (256 thr) | + working (est.) | regs/thread | regs/block | reg-allowed blocks/SM |
|
||||
|---|---|---|---|---|---|---|
|
||||
| 128 (no slab) | 16384 | 64 | ~50 | ~114 | ~29 K | 2 (tight) |
|
||||
| 64 | 8192 | 32 | ~50 | ~82 | ~21 K | 3 |
|
||||
| 32 | 4096 | 16 | ~50 | ~66 | ~17 K | 3 |
|
||||
|
||||
So on registers alone, dv_tile≤64 admits ≥2 blocks/SM. **Shared memory is then the binding cap**, and it's governed by **C**, not dv_tile (Kc/Qc/A all scale with C, only U scales with dv_tile):
|
||||
|
||||
| Config | Kc+Qc (bf16) | A/P (f32) | U (f32) | single | +cp.async dbl-buf K/Q | blocks/SM (shared) |
|
||||
|---|---|---|---|---|---|---|
|
||||
| C=64, dv_tile=128 | 32 KB | 16 KB | 32 KB | 80 KB | 112 KB ✗(no room!) | **1** |
|
||||
| C=64, dv_tile=64 | 32 KB | 16 KB | 16 KB | 64 KB | 96 KB ✓ | **1** |
|
||||
| **C=32, dv_tile=64** | 16 KB | 4 KB | 8 KB | **28 KB** | **44 KB ✓** | **2** |
|
||||
| C=32, dv_tile=32 | 16 KB | 4 KB | 4 KB | 24 KB | 40 KB ✓ | **2** |
|
||||
|
||||
**Finding the scope doc missed:** C=64-no-slab is shared-saturated at 80 KB — there is **no room for cp.async double-buffering**, so the 1-block/SM kernel would have *no latency hiding* and likely still lose. C=64 needs dv_tile≤64 *just to make room for cp.async*, and is still 1 block/SM. **Genuine 2 blocks/SM requires C=32** (to drop Kc/Qc/A under the 49.5 KB/block budget).
|
||||
|
||||
## 4. cp.async double-buffering (depth 2, no TMA)
|
||||
|
||||
At 1 block/SM (C=64 path) cp.async is the *only* latency-hiding mechanism, so it's mandatory, not optional. Plain Ampere `cp.async` (`cp.async.commit_group` / `cp.async.wait_group`) — **no `cp.async.bulk`/TMA on sm_121.** Stage the **next chunk's Kc, Qc** (and Vc if the KL-gate doesn't force V from global) into a second shared buffer while the current chunk's mma runs. Depth **2 only** — the sibling GEMM kernel proved multistage saturates BW past depth 2. The double-buffer cost is already in the "+cp.async" column above (44 KB at C=32 keeps 2 blocks/SM).
|
||||
|
||||
## 5. Launch config (concrete) + honest occupancy estimate
|
||||
|
||||
**Recommended default (batched-prefill serving regime, n_seqs≥2):**
|
||||
```
|
||||
C = 32 ; dv_tile = 64 ; n_slabs = 2 ; WARPS = 8
|
||||
grid = dim3(H=32, n_seqs, 2)
|
||||
block = dim3(256, 1, 1)
|
||||
smem = 44 KB (Kc/Qc bf16 ×2 dbl-buf + A/P f32 + U f32) // cudaFuncSetAttribute return CHECKED (0031 precedent)
|
||||
→ 2 blocks/SM. n_seqs≥2 ⇒ ≥128 blocks ⇒ 48/48 SMs at full 2-block occupancy (100%), 1.33 waves.
|
||||
A/Gram/solve recomputed 2× across slabs (state-update per slab is 2× the A work ⇒ ~25% redundant-flop overhead).
|
||||
```
|
||||
|
||||
**Single-stream prefill (n_seqs=1) saturator:**
|
||||
```
|
||||
C = 32 ; dv_tile = 32 ; n_slabs = 4 ; WARPS = 8
|
||||
grid = dim3(32, 1, 4) = 128 blocks ⇒ 2 blocks/SM on all 48 SMs (100%) even at n_seqs=1.
|
||||
Cost: A recomputed 4×, and at dv_tile=32 the A bucket ≈ the per-slab state bucket ⇒ ~1.5-2× total-flop overhead.
|
||||
```
|
||||
|
||||
**BW-max alternative (1 block/SM, bench against the above):**
|
||||
```
|
||||
C = 64 ; dv_tile = 64 ; n_slabs = 2 ; WARPS = 8 ; smem = 96 KB (dbl-buf, fits 99 KB)
|
||||
→ 1 block/SM, but 4× state-BW cut (vs 2× at C=32) + grid ×2. n_seqs=1 ⇒ 64 blocks ⇒ 48/48 SMs busy (67% warp-occ).
|
||||
```
|
||||
|
||||
**Occupancy summary:**
|
||||
|
||||
| Config | blocks/SM | regs/thread | smem/block | SM util @ n_seqs=1 | SM util @ n_seqs≥2 | state-BW cut | redundant-A |
|
||||
|---|---|---|---|---|---|---|---|
|
||||
| 0031 | 1 | scalar | 89 KB | 32/48 busy (starved) | 1024 blk, no overlap | 1× (C=16) | none |
|
||||
| C=32 dv64 (default) | **2** | ~82 | 44 KB | 48 busy, 67% occ | **100%** | 2× | 2× (~25%) |
|
||||
| C=32 dv32 (1-seq) | **2** | ~66 | 40 KB | **100%** | 100% | 2× | 4× (~1.5-2×) |
|
||||
| C=64 dv64 (BW-max) | 1 | ~114 | 96 KB | 48 busy, 67% occ | 100%, multi-wave | **4×** | 2× |
|
||||
|
||||
The C=32 (occupancy) vs C=64 (BW) choice is the empirical fork the scope doc defers to Phase-3 bench: 2 blocks/SM at half the BW saving, vs 1 block/SM at full BW saving + cp.async. **Wire both behind the existing `GDN_CHUNK_MIN` gate plus a `GDN_CHUNK_C` / `GDN_DV_TILE` selector and A/B them; do not assume.**
|
||||
|
||||
## 6. Residual risk — register pressure likely caps it at 1 block/SM (honest)
|
||||
|
||||
The ≥2-blocks/SM result rests on the **~50 working-regs/thread estimate**, which is optimistic:
|
||||
|
||||
- **The blocked-forward-subst A⁻¹ (step 7) is the swing factor.** The in-register 16×16 unit-lower-triangular diagonal inverse + the off-diagonal mma coupling + mma operand fragments + the **accumulator→operand fragment transpose** for reusing the register-resident S as a step-3/4 operand (a `movmatrix`/shared round-trip, since S lives in C-fragment layout but steps 3/4 need it as an A/B operand) can push working regs to **80-120**. At 256 threads, regs/thread > 128 ⇒ > 32 K regs/block ⇒ **silently drops to 1 block/SM** regardless of the 44 KB shared headroom. The scope doc names this exactly: "blocked-forward-subst register pressure trades against state-register pressure; both compete for the same budget."
|
||||
- **Mitigation ladder, in order:** (i) 12 warps to dilute (dv_tile=64: state 32→21 regs/thread); (ii) `__launch_bounds__(256, 2)` to force the compiler under 128 regs/thread (risks spills to local → BW back); (iii) smaller dv_tile (more grid, more redundant A). If all fail, accept **1 block/SM and lean on cp.async double-buffering + the 4× BW cut + mma throughput** — which is still very likely a win over 0031's serial-f32/-22%, just not the 2-block target.
|
||||
- **Grid-starvation at n_seqs=1 is structural** (H=32 < 96): only dv_tile=32 fully saturates a single stream, and it pays ~1.5-2× redundant-A flops. This is an inherent floor of the chunked formulation on a 32-head layer; batched serving (n_seqs≥2) is where the design cleanly saturates.
|
||||
- **Precision↔occupancy coupling:** the 2-block budget assumes **bf16 K/Q** staging. If the KL-gate demands tf32 for the KS/QS S0-products (decay-coupled), that needs a second 32-bit K/Q copy or 3×tf32 — both inflate shared/registers and can knock C=32 back toward 1 block/SM. The occupancy win is contingent on bf16 Gram clearing the gate.
|
||||
|
||||
**Bottom line:** 2 blocks/SM and full 48-SM saturation are *reachable* — via **C=32 + dv-slab (64 for serving, 32 for single-stream) + 8 warps + depth-2 cp.async** — on the **shared-memory** budget. Whether the **register** budget also permits it is the one genuinely open risk, and the A⁻¹ solve's footprint is what decides it; the realistic outcome remains the scope doc's honest **1-2 blocks/SM**, with the win carried by mma + cp.async + the C× BW cut rather than by high occupancy.
|
||||
|
||||
Files: design lives in `backend/cpp/llama-cpp-localai-paged/docs/TENSORCORE_GDN_SCOPE.md` (§3 occupancy plan — this analysis sharpens it); kernel/dispatcher to modify `backend/cpp/llama-cpp-paged-dev/ggml/src/ggml-cuda/gated_delta_net.cu` (grid at :184, chunked launch added by patch `patches/paged/0031-paged-chunked-gdn-prefill-scan-kernel.patch`); tf32 tile at `ggml/src/ggml-cuda/mma.cuh:1089`; H source `src/llama-model.cpp:504`.
|
||||
|
||||
## 5. Synthesized build plan + milestones + gate
|
||||
|
||||
Historical note: this plan predates the shipped f32-only M5 tensor-core GDN
|
||||
path in patch `0047`. Current code parses `GDN_CHUNK_MIN` and `GDN_TC`; the
|
||||
older `GDN_CHUNK_OFF` and `GDN_CHUNK_TC` names in this section are obsolete.
|
||||
Phase71 revalidated the current default against sequential-disabled and
|
||||
serial-chunked modes on DGX and kept M5 as shipped. Use this document as
|
||||
background for any larger FLA/CuteDSL-class redesign, not as the active next
|
||||
patch queue.
|
||||
|
||||
All anchors were verified at the time of writing. 0031's kernel body, the
|
||||
7-step structure, the `GDN_CHUNK_MIN` gating at the `if constexpr (!KDA &&
|
||||
!keep_rs_t)` site, the `launch_gdn_chunked<128,16>` template, the smem formula,
|
||||
and the test-backend-ops shapes were confirmed. The scope doc's KL gate,
|
||||
3xtf32 ladder, risk register, and Phase 0-3 plan were confirmed. Here is the
|
||||
historical build-ready synthesis.
|
||||
|
||||
---
|
||||
|
||||
# BUILD-READY PLAN: tensor-core chunked-GDN prefill kernel (sm_121a)
|
||||
|
||||
Anchors (absolute): kernel + launch + gate live in `ggml/src/ggml-cuda/gated_delta_net.cu` (grid at :184), added by `/home/mudler/_git/LocalAI/.claude/worktrees/feat+paged-attention/backend/cpp/llama-cpp-localai-paged/patches/paged/0031-paged-chunked-gdn-prefill-scan-kernel.patch`. tf32 tile `mma(tile<16,8,float>&D, tile<16,8,float>&A, tile<8,8,float>&B)` = `mma.sync.aligned.m16n8k8.row.col.f32.tf32.tf32.f32` in `ggml/src/ggml-cuda/mma.cuh` (m16n8k8 overload ~976-984, dispatch ~1089), gated by `AMPERE_MMA_AVAILABLE`. PAD/cp.async patterns from `patches/paged/0034-feat-paged-native-NVFP4-W4A4-FP4-MMA-large-M-prefill.patch`. Gate/precedent docs: `docs/TENSORCORE_GDN_SCOPE.md`, `docs/PAGED_BITEXACT_NOTE.md`, `README.md` s5. Microbench: `~/scratch_tc_gdn_poc/gdn_gram_bench.cu` (DGX). Last patch in series is 0042 → this work is patches 0043+.
|
||||
|
||||
The new kernel is `gated_delta_net_chunked_tc_cuda<S_v, C, DV_TILE>`, a sibling to 0031's `gated_delta_net_chunked_cuda`. Symbols below reuse 0031's smem names (`Sd, Kc, Qc, Ud, Amat, csh, gam, bet`).
|
||||
|
||||
---
|
||||
|
||||
## (1) Phase-by-phase kernel structure
|
||||
|
||||
Block = **256 threads / 8 warps** in a **4×2 (WM×WN)** warp grid. State `S` (`dk×dv_tile`) is **register-resident in the step-6 accumulator layout** (`tile<16,8,float>` grid). Grid = `dim3(H, n_seqs, n_slabs)`, `blockIdx.z` = dv-slab. Chunk axis is the serial recurrence (NOT a grid axis). Invariant preserved from 0031: read pre-update `S0` (P3/P4) → solve → output (P5) → **overwrite S last** (P6). Single accumulator, no state double-buffer.
|
||||
|
||||
Per chunk `c0` (the loop body):
|
||||
|
||||
**Phase A - chunk load + gate prefix (f32, cooperative).** Load `Kc[C][dk]`, `Qc[C][dk]` **as bf16** (tf32 K/Q blows the 99KB budget - see §5 of the state design), load `V` chunk. Compute `csh = cumsum(g)` (≤0), `gam = exp(csh)` (≤1), `bet` - all f32, identical to 0031 lines (the `j==0` prefix scan, kept scalar; it is <1KB and hides under the Grams). cp.async depth-2 prefetch of the *next* chunk's `Kc/Qc` starts here.
|
||||
|
||||
**Phase B - state restage (accumulator→B bridge).** The crux. `S0` lives as P6's D/accumulator fragments but P3/P4 need it as a **B operand** (`tile<8,8>`, K-major over `i`). Bounce the `dk×dv_tile` state through a transient smem tile that **overlays the `Ud∪Amat` region** (free at chunk entry - U/A not yet computed) → `load_generic` back as B fragments (NOT `ldmatrix`: it is `.b16`-only, unusable for tf32; use `load_generic`). Paid `n_tokens/C` times, **0KB net persistent smem**. KS and QS share this one restage.
|
||||
|
||||
**Phase C - Gram + state-boundary products (the matmuls that read pre-update S0).**
|
||||
- **P1 `KK→A`** = `Kc·Kcᵀ`, M=C N=C K=dk, lower-tri (~½ tiles). **tf32-safe** (PoC-proven NMSE ~3e-9). Apply `A = I + tril(βₜ·d(t',t)·KK, -1)` in **f32** after the mma.
|
||||
- **P3 `KS`** = `Kc·S0`, M=C N=dv K=dk. **3xtf32** (state-boundary, feeds the solve). Output → `Ud` region (becomes RHS).
|
||||
- **P4 `QS`** = `Qc·S0`, M=C N=dv K=dk, **fused with P3 on the shared S0 B-fragments**. **3xtf32** (γ-attenuated → first demote candidate). Seed the **O accumulator fragments register-resident with `γₜ·QS`** immediately (avoids parking QS in smem through to Phase F). Restage overlay is now free; `Ud`/`Amat` reclaim it.
|
||||
|
||||
**Phase D - A-inverse (form T = A⁻¹ explicitly, then wide apply).**
|
||||
- **Phase-D inverses:** 4 diagonal `16×16` unit-lower-tri blocks, **f32 in shared-memory column-parallel forward substitution** (thread `c` solves `A_ii x = e_c`). No tensor cores, no reduced precision (this is the strong-coupling amplifier). 4 blocks on 4 warps in parallel, hides entirely under the Phase-C/RHS Grams.
|
||||
- **Phase-O off-diagonal:** wavefront (anti-diagonal) schedule, critical path `n_b-1=3` not 6. For each i>j: `P_ij = Σ_m A_im·T_mj` then `T_ij = -T_ii·P_ij`, on `m16n8k8`. **3xtf32 default-on** (~64 tiny mma total, negligible). `T` overwrites the `A` scratch in place.
|
||||
|
||||
**Phase E - RHS + apply.** `RHS = βₜ(vₜ - γₜ·KS)` in **f32** (uses P3 result + V) → `Ud`. **`U = T·RHS`** as one dependency-free wide **tf32** GEMM, M=C N=dv K=C (the bulk, 128 mma/warp at full dv), in place → `Ud`.
|
||||
|
||||
**Phase F - intra-chunk output.**
|
||||
- **P2 `QK→P`** = `Qc·Kcᵀ`, reuse `Amat` (now free after T consumed). **tf32-safe**. Apply `P = d(t',t)·QK` in **f32** (bounded, decay pre-baked - preserves the bounded de-gating invariant).
|
||||
- **P5 `O += P·U`**, M=C N=dv K=C, P lower-tri (~½ tiles). **tf32-safe** (P f32-bounded first). Accumulate into the O fragments already seeded with `γₜ·QS`. Write `O*scale` to `dst`.
|
||||
|
||||
**Phase G - state carry (overwrites S0 last).** `DU = d(t,last)·U` in f32. **Scale the persistent S accumulator fragments by `γ_last` in f32 in-register first**, then **P6 `S_C += Kcᵀ·DU`** = `Kcᵀ·DU`, M=dk N=dv K=C, **3xtf32 (the strongest ladder candidate - compounds over every chunk)**, accumulated straight into the persistent registers. `Kc` is read **transposed** here (second fragment view, `load_generic` transpose). No restage-out: S stays resident for the next chunk.
|
||||
|
||||
After the loop: final-state write-back (M-layout), identical to 0031's tail.
|
||||
|
||||
Buffer lifecycle (single `Amat`, single `Ud`, as 0031): `Amat`: A(P1) → T(Phase-D/O, in place) → consumed by apply → P(P2) → consumed by P5. `Ud`: KS(P3) → RHS(Phase-E) → U(apply, in place) → read by P5 (B) and P6 (B, scaled to DU). Restage tile overlays `Ud∪Amat` only at chunk entry (Phase B), before either is written.
|
||||
|
||||
---
|
||||
|
||||
## (2) Build sequence - incremental, each independently GPU-verifiable vs 0031
|
||||
|
||||
Each milestone is a **separate patch** stacked on 0031, **green on `test-backend-ops GATED_DELTA_NET` + greedy-md5 stable before the next is started**. Reference for every step = the `test_gated_delta_net` op's f64/CPU oracle (already in-tree) and 0031's serial-chunked output. **No milestone integrates on top of an unverified one.**
|
||||
|
||||
| M | Scope | Patch | GPU verification gate (vs 0031 / op oracle) |
|
||||
|---|---|---|---|
|
||||
| **M0** | Re-confirm regime, NO code (scope Phase 0) | - | Profile 0031 (`GDN_CHUNK_MIN` low): confirm GDN prefill bucket dominates + grid-starved at low n_seqs. If not, kill the lever now. |
|
||||
| **M1** | **DGX microbench (NO kernel yet)** - extend `gdn_gram_bench.cu` with KS/QS/PU/KᵀU microkernels + Phase-D/O T-formation + T·RHS apply, each with f64 host oracle measuring **κ(A)** and tf32-vs-3xtf32 NMSE per rung, incl. adversarial `g∈[-20,-1e-4]` | - | **The cheap go/no-go before multi-week kernel work.** Pass = default precision config (f32 diag + 3xtf32 off-diag + tf32 bulk) reaches ~PoC `3e-9`-grade on benign data and survives the κ(A) weak-decay corner within the ladder. Mirrors the PoC that proved 6.7×→9.3×. |
|
||||
| **M2** | In-kernel: replace **only** step-1/2 serial Grams (KK/QK) with tensor-core tiles. **C=16, scalar everything else, same occupancy** (scope Phase 1 / PoC integration) | 0043 | test-backend-ops 128-shapes green via KL gate (NMSE if it passes); greedy-md5 stable. |
|
||||
| **M3** | Add **P3/P4 (KS/QS)** tensor-core (3xtf32) + S restage bridge. Still C=16, scalar solve + scalar O/state | 0044 | Same gate. Isolates the accumulator→B bridge correctness. |
|
||||
| **M4** | **A-inverse** Phase-D (f32 diag) + Phase-O (3xtf32 off-diag), form T; replace 0031's serial fwd-subst. Still C=16 | 0045 | Same gate + the adversarial-decay op case (this is the amplifier). |
|
||||
| **M5** | **Apply `U=T·RHS`** + **P5 `P·U`** tensor-core. Still C=16 | 0046 | Same gate. |
|
||||
| **M6** | **P6 `Kᵀ(D·U)`** tensor-core + **register-resident state** (step-6 accumulator layout) + accumulator→B restage in steady state. State leaves smem here | 0047 | Same gate. Frees the 64KB that forced C=16. |
|
||||
| **M7** | **Flip C=16→C=64, full dv (CONFIG A ~86KB, 1 blk/SM)**, 8-warp 4×2 grid, PAD=4 smem | 0048 | Gate + **first A/B bench vs sequential** (S_PP at n_seqs≥2). |
|
||||
| **M8** | **Occupancy:** C=32 + dv-slab grid `(H,n_seqs,n_slabs)` (CONFIG C, 2 blk/SM) + cp.async depth-2; selectors `GDN_CHUNK_C`/`GDN_DV_TILE` | 0049 | Gate + A/B bench across {C=32/dv64, C=32/dv32, C=64/dv64-BW-max}; pick winner per regime. |
|
||||
|
||||
---
|
||||
|
||||
## (3) Bit-exact / KL gate plan
|
||||
|
||||
**md5 is per-path and will NOT match** 0031-serial or the sequential recurrence (different FP reduction order). This is the established `-paged` precedent (`PAGED_BITEXACT_NOTE.md`): per-path md5, validated benign. So:
|
||||
|
||||
- **Binding gate = KL** (not strict NMSE): `KLD(tensorcore ‖ f16) ≤ KLD(sequential ‖ f16)` plus a PPL band, on the README s5 harness. NMSE is *expected to fail* at reduced precision (new path on a new path); NMSE-pass is a bonus, KL-pass is the bar.
|
||||
- **Stability gate:** greedy-md5 **stable across runs** (deterministic), not equal to the serial path.
|
||||
- **Adversarial op case mandatory:** `g∈[-20,-1e-4]` (the dangerous middle-decay regime where κ(A) grows); strong-decay underflows to 0 (safe), weak-decay is well-conditioned (tf32's 8-bit exponent holds γ range), the middle is the only empirical risk.
|
||||
|
||||
**Precision default config (the bet that clears the gate):** f32 diagonal inverse (mandatory, already f32) · **3xtf32 off-diagonal coupling** (default-on, negligible ~64-mma cost) · **tf32** Grams + apply · **bf16** K/Q staging (well-conditioned KK/QK only) · decays/γ/β **always f32 outside the mma** (invariant, not a rung). Hold **P6 state carry at 3xtf32 longest** (it compounds over every chunk).
|
||||
|
||||
**3xtf32 ladder (cheapest→dearest) if default misses the gate:** (1) KK Gram→3xtf32; (2) apply **block-diagonal `T_ii·RHS_i`**→3xtf32 (within-window strong coupling, mixed-precision-by-distance); (3) +**δ=1 off-diagonal** apply→3xtf32 (block-boundary adjacent pairs e.g. tokens 15↔16); (4) **full apply**→3xtf32 (≈+2× apply, expensive escape); (5) KS/QS→3xtf32; (6) fall back to direct blocked back-substitution in 3xtf32, else keep 0031's serial path. **Demote order if the gate has margin:** P4→P3, holding P6 at 3xtf32. If even all-3xtf32 misses, the residual is the f32 diagonal solve (already f32) → not fixable by more mma precision → fall to (6). Record the final rung in `PAGED_BITEXACT_NOTE.md` + README s5.
|
||||
|
||||
---
|
||||
|
||||
## (4) Slot into 0031's existing framework (historical, superseded by 0047)
|
||||
|
||||
Same dispatch site - the `if constexpr (!KDA && !keep_rs_t)` block inside `launch_gated_delta_net` (0031 patch, after `init_fastdiv_values`). Extend, don't replace:
|
||||
|
||||
- Current code keeps `GDN_CHUNK_MIN` as the token threshold and uses `GDN_TC`
|
||||
as the tensor-core level selector. It does not parse `GDN_CHUNK_OFF` or
|
||||
`GDN_CHUNK_TC`.
|
||||
- Historical plan: add **`GDN_CHUNK_TC`** selector: `0` = 0031 serial-solve chunked (fallback, retained), `1` = tensor-core. Add **`GDN_CHUNK_C` ∈ {16,32,64}** and **`GDN_DV_TILE` ∈ {32,64,128}** for A/B; defaults `C=32, DV_TILE=64` (CONFIG C) for serving, `DV_TILE=32` saturator for n_seqs=1.
|
||||
- New launcher `launch_gdn_chunked_tc<128, C, DV_TILE>` mirrors `launch_gdn_chunked`: `cudaFuncSetAttribute(...MaxDynamicSharedMemorySize...)` **return-checked** (0031 precedent), `grid = dim3(H, n_seqs, n_slabs)`, `block = dim3(256,1,1)`. Per-slab the kernel recomputes A/A⁻¹/gates (dv-independent), dv-slices S/Ud/O.
|
||||
- **Default OFF** (`gdn_chunk_min=INT_MAX`) exactly as 0031 ships. Flip the default to on **only when** the M8 A/B shows an S_PP win over the tuned sequential recurrence at the serving regime (n_seqs≥2) **and** the KL gate + adversarial op case hold - recorded in README s5 (dev notes / rejected-flat levers) and `PAGED_BITEXACT_NOTE.md`. Until then it ships like 0031: opt-in, regression-free default.
|
||||
- Extend the test-backend-ops block 0031 added (the `S_v==128` shapes at lines after :9398) so the tc path is exercised at C=64 and C=32 in CI.
|
||||
- New per-path md5 acknowledged in the dispatch comment (tc-md5 ≠ serial-chunked-md5 ≠ sequential-md5; all benign, KL-validated).
|
||||
|
||||
---
|
||||
|
||||
## (5) Top 3 risks that could make it NOT beat sequential + kill-criteria
|
||||
|
||||
**Risk 1 - Register pressure forces 1 block/SM (the swing factor).** The ~50 working-regs/thread estimate is optimistic; the A⁻¹ blocked solve (in-register `16×16` diag inverse), the accumulator→B restage transpose, and the O+state transient accumulators can push working regs to 80-120. At 256 threads, >128 regs/thread → >32K regs/block → **silently 1 block/SM regardless of the 44KB shared headroom**, and local-memory spills push BW back. *Mitigation ladder:* (i) 12 warps (dilute state 32→21 regs/thread); (ii) `__launch_bounds__(256,2)`; (iii) smaller `DV_TILE`. **Kill criterion:** if after the full ladder the M8 occupancy build still spills to local OR stays 1 block/SM, **and** the CONFIG-A BW-max 1-block path (C=64, dv64, 96KB, cp.async, 4× state-BW cut) **also** fails to beat sequential S_PP at n_seqs≥2 in the A/B bench → the occupancy lever is dead; keep 0031 serial-chunked behind `GDN_CHUNK_TC=0`, record rejected in README s5.
|
||||
|
||||
**Risk 2 - Precision: tf32 (and even all-3xtf32) misses the KL gate in the weak-decay/aligned-keys κ(A) corner.** The inverse amplifies error; κ(A) is data-dependent and grows where keys align and decay is weak. **Detected cheaply at M1** (microbench measures κ(A) + per-rung NMSE on the adversarial case *before* the kernel exists). **Kill criterion:** if at M1 the **top of the ladder (all-3xtf32 + f32 diagonal)** cannot reach f32-grade on `g∈[-20,-1e-4]`, OR at M4+ `KLD(tc‖f16) ≤ KLD(seq‖f16)` fails on that op case at the top rung → the tensor-core solve is not numerically viable as a default; fall to ladder rung (6) (direct back-subst 3xtf32); if that also misses, abandon the tc solve and keep 0031 serial. **Fail-fast:** M1 gates this before any multi-week kernel commitment.
|
||||
|
||||
**Risk 3 - Grid starvation at n_seqs=1 is structural (H=32 < the ~96 blocks needed for 2 blk/SM × 48 SM).** Only `DV_TILE=32` (4 slabs) fully saturates a single stream, and it pays ~1.5-2× redundant-A flops (A/A⁻¹/gates recomputed per slab) plus the per-chunk restage. **Kill criterion:** if the M8 bench shows single-stream (n_seqs=1) S_PP is slower than sequential even at full saturation (dv32×4) due to redundant-A + restage overhead, **and** the batched regime (n_seqs≥2) gain also fails to materialize → the lever only helps a regime the target workload doesn't hit → keep default-OFF, ship as opt-in experiment only, record. (If n_seqs≥2 *does* win, ship enabled for the serving regime and gate single-stream back to sequential via `GDN_CHUNK_MIN` + an n_seqs check - a partial, honest win.)
|
||||
|
||||
**Overarching kill gate:** the disposition is the bench, not the theory. The kernel flips to default-on only when it beats the tuned sequential recurrence at the serving regime AND clears the KL + adversarial gates. Any milestone that regresses test-backend-ops or md5-stability halts the stack until fixed; M1 and M0 are the cheap fail-fast exits before the expensive kernel work.
|
||||
@@ -1,362 +0,0 @@
|
||||
# TENSORCORE_GDN_SCOPE - tensor-core chunked gated-DeltaNet prefill (design only)
|
||||
|
||||
**Status: DESIGN + SCOPE ONLY. No kernel written, no GPU run, no PTX in this pass.**
|
||||
This scopes the follow-up recorded by patch 0031 and README section 5: a
|
||||
tensor-core (`mma`) chunked gated-DeltaNet (GDN) prefill kernel - the path that
|
||||
would actually *beat* the tuned sequential scan and close the GDN prefill bucket
|
||||
toward vLLM. vLLM's chunked GDN scan was measured ~2.5x cheaper in the prefill
|
||||
ground-truth precisely because it pushes the intra-chunk products through
|
||||
tensor-core matmuls; patch 0031 proved the chunking math but, with serial
|
||||
per-thread reductions at the GB10-forced `C=16`, came out ~22% *slower* than the
|
||||
sequential recurrence. This document scopes replacing those reductions with
|
||||
`mma.sync` matmuls and lifting the occupancy ceiling.
|
||||
|
||||
> **Read patch 0031 + README section 5 first.** The bounded/stable de-gating form
|
||||
> (pairwise decays `d <= 1`, `gamma <= 1`), the per-path bit-exact precedent, and
|
||||
> the honest negative ("C=16 all-shared -> 1 block/SM -> serial reductions -> 22%
|
||||
> slower, grid-starved at low n_seqs") are the starting point. This doc does not
|
||||
> re-derive the algebra; it maps it onto tensor cores.
|
||||
|
||||
> **Regime note (the mechanism, read this).** The sequential scan is
|
||||
> **bandwidth-bound**: it re-streams the entire `128x128` f32 state (64KB) once
|
||||
> *per token*. README section 5 already records it runs at ~84.7% of GB10 peak BW
|
||||
> (decode) and the recurrence is a llama *win* vs vLLM's BW. So a tensor-core
|
||||
> kernel does **not** win by doing the same work faster - it wins by **changing
|
||||
> the work**: chunking by `C` reads/writes the state `n_tokens/C` times instead of
|
||||
> `n_tokens` (a ~`C`x cut in state traffic, the dominant prefill GDN cost), and the
|
||||
> price is `O(C^2)` extra intra-chunk dot-products per chunk. The naive 0031 paid
|
||||
> that price in serial f32 reductions, which cost *more* than the BW it saved -
|
||||
> hence 22% slower. **Tensor cores make the added intra-chunk flops nearly free,
|
||||
> so the BW saving becomes a net win.** That is exactly why vLLM's chunked scan is
|
||||
> 2.5x cheaper. The whole lever rests on this trade; if a GPU re-profile shows
|
||||
> prefill GDN is *not* state-BW-bound, stop and re-scope (step 0 below).
|
||||
|
||||
---
|
||||
|
||||
## 1. GB10 tensor-core reality (sm_121a) - confirmed, not assumed
|
||||
|
||||
GB10 / DGX Spark reports **compute capability 12.1 (sm_121)**, CUDA 13 (README
|
||||
section "Hardware: GB10 / DGX Spark (CUDA 13, sm_121)"). sm_121a is **consumer
|
||||
Blackwell** (the SM12x family, same tensor-core programming model as RTX 50 /
|
||||
sm_120), **not** data-center Blackwell (sm_100a / GB200). This distinction is the
|
||||
single most important input to the design and is confirmed from sources, not
|
||||
assumed:
|
||||
|
||||
- **No `wgmma`.** Warp-group MMA is Hopper (sm_90a) only; targeting SM12x yields
|
||||
`ptxas error: Instruction 'wgmma.fence' not supported on .target 'sm_120'`.
|
||||
Do **not** design around Hopper-style warp-group MMA.
|
||||
- **No `tcgen05` / no TMEM.** SM12x lacks the Tensor Memory hardware entirely, so
|
||||
the autonomous 5th-gen tensor-core path (`tcgen05.mma`, the sm_100a data-center
|
||||
instruction) is unavailable. This is the same wall that makes vLLM/CUTLASS fall
|
||||
back to Marlin and gate FP4 to sm_100a on GB10 (tracked in CUTLASS #2800/#2947,
|
||||
vLLM #43906). We cannot use it either.
|
||||
- **What sm_121a DOES have: extended `mma.sync`.** The Ampere/Ada warp-level
|
||||
`mma.sync` family, extended with the Blackwell numeric formats (FP8/FP6/FP4).
|
||||
"Consumer Blackwell put new data types on top of the oldest programming model."
|
||||
For our operands (q/k/v/state are f32 in the op, see below) the usable tiles are
|
||||
the standard warp-level ones:
|
||||
- **bf16/f16 inputs, f32 accumulate:** `mma.sync.aligned.m16n8k16` (and
|
||||
`m16n8k8`). 7-bit (bf16) / 10-bit (f16) input mantissa.
|
||||
- **tf32 inputs, f32 accumulate:** `m16n8k8` / `m16n8k4`. 10-bit input mantissa
|
||||
- the **highest-precision tensor-core option** on this part, and the one this
|
||||
design defaults to (the GDN is decay-sensitive; see section 4).
|
||||
- FP8 (`m16n8k32`) / FP4 (`m16n8k64.kind::mxf4nvf4`, block-scaled) compile on
|
||||
sm_121a but are **out of scope** here - the GDN q/k/v/state are not 4/8-bit.
|
||||
- **`cp.async` is available** (Ampere+), so global->shared double-buffering of the
|
||||
K/Q chunk tiles is on the table for the occupancy phase. There is **no TMA** on
|
||||
SM12x; staging is plain `cp.async`, not `cp.async.bulk`.
|
||||
|
||||
**Reuse, do not hand-roll PTX.** ggml already ships a warp-level MMA tile
|
||||
abstraction at `ggml/src/ggml-cuda/mma.cuh` (the `tile<M,N,T>` fragments +
|
||||
`mma()` used by the FlashAttention-mma and MMQ kernels), and it already routes
|
||||
through `turing_mma_available(cc)` / `ampere_mma_available(cc)` - i.e. it is
|
||||
sm_121-correct today. Build the GDN matmuls on that API (bf16/half/tf32 fragments,
|
||||
f32 accumulators), not on raw `asm volatile("mma.sync...")`. This de-risks the
|
||||
kernel and keeps it consistent with the backend's other tensor-core paths.
|
||||
|
||||
**Bottom line for the design:** the kernel is a **warp-synchronous `mma.sync`**
|
||||
kernel (Ampere-class programming model with Blackwell silicon), *not* a
|
||||
warp-group / TMA / tcgen05 kernel. Every "wgmma"/"tcgen05" idea from FLA's
|
||||
sm_90/sm_100 kernels must be down-translated to `mma.sync` + `cp.async`. Patch
|
||||
0031's and README's shorthand "mma/wgmma" should be read as **mma only** on this
|
||||
part.
|
||||
|
||||
---
|
||||
|
||||
## 2. Mapping the chunked GDN matmuls onto `mma.sync`
|
||||
|
||||
The chunked gated-delta-rule (patch 0031 header) has six dot-product families.
|
||||
Five are plain matmuls and map cleanly to `mma`; the sixth (the A-inverse) is a
|
||||
unit-lower-triangular solve and is the one subtle case. Notation: `C` = chunk
|
||||
length, `dk = dv = S_v = 128` (GDN head dim), per `(head, seq)` block.
|
||||
|
||||
| # | Product (0031 step) | Shape | mma form | Notes |
|
||||
|---|---|---|---|---|
|
||||
| 1 | `KK[t,t'] = k_t . k_t'` (for `A`) | `C x C` over `k=dk=128` | `(C x dk) x (dk x C)` | Gram matrix; only strict-lower triangle used. Decay `d(t',t)` + `beta_t` applied **after** mma in f32. |
|
||||
| 2 | `QK[t,t'] = q_t . k_t'` (for `P`/`O`) | `C x C` over `k=dk` | `(C x dk) x (dk x C)` | Lower triangle (`t' <= t`); decay applied after in f32. |
|
||||
| 3 | `KS[t,j] = (S0^T k_t)[j]` | `C x dv` over `k=dk` | `(C x dk) x (dk x dv)` | `S0` is the chunk-entry state (stationary operand). Feeds RHS of the solve. |
|
||||
| 4 | `QS[t,j] = (S0^T q_t)[j]` | `C x dv` over `k=dk` | `(C x dk) x (dk x dv)` | The `gamma_t` cross-chunk term of `O`. |
|
||||
| 5 | `O += P . U` | `C x dv` over `k=C` | `(C x C) x (C x dv)` | `P` (decay-masked `QK`) times the solved `U`. |
|
||||
| 6 | `S_C += K^T (D .* U)` | `dk x dv` over `k=C` | `(dk x C) x (C x dv)` | The state update; `D` = `diag(d(t,last))` applied to `U` in f32 first. |
|
||||
| 7 | `U = A^{-1} RHS` | `C x C` solve, `C x dv` RHS | blocked fwd-subst (see below) | The only non-GEMM. |
|
||||
|
||||
**Critical precision invariant (preserve the bounded de-gating).** Every decay
|
||||
(`gamma_t`, `d(t',t) = exp(cs_t - cs_t')`) and every `beta_t` stays in **f32** and
|
||||
is applied as an elementwise scale **before/after** the mma, never inside it. The
|
||||
mma only ever multiplies the raw, unweighted dot-products (`k.k`, `q.k`,
|
||||
`S0^T k`, `S0^T q`, `P.U`, `K^T U`). This keeps the strong-decay underflow-to-zero
|
||||
behaviour (the adversarial `g in [-20, -1e-4]` op test) exactly as 0031 has it -
|
||||
the numerically delicate part never touches reduced precision. This is the
|
||||
discipline that makes a tf32/bf16 mma kernel safe for a decay-sensitive op.
|
||||
|
||||
### The A-inverse (step 7) - it CAN be tensor-core'd
|
||||
|
||||
`A = I + N`, `N = tril(beta_t d(t',t) k_t.k_t', -1)` is **strictly lower
|
||||
triangular**, hence **nilpotent** (`N^C = 0`). Two routes, both better than 0031's
|
||||
serial per-thread forward substitution:
|
||||
|
||||
- **Blocked forward substitution (RECOMMENDED, this is the FLA "UT transform").**
|
||||
Partition `C` into sub-blocks of `b` (e.g. `b = 16`, one mma `m`-tile). Invert
|
||||
each `b x b` diagonal block in registers (it is unit-lower-triangular `b x b`,
|
||||
cheap: a short serial solve or the finite Neumann series on a `b`-nilpotent,
|
||||
`<= b-1` terms), then propagate to the off-diagonal sub-blocks with **mma**
|
||||
(the inter-block coupling `U_i -= A_ij U_j` is exactly a `(b x b) x (b x dv)`
|
||||
matmul). For `C = 64, b = 16` that is 4 tiny in-register diagonal solves + a
|
||||
triangular sweep of mma updates - the bulk of the solve is on tensor cores, only
|
||||
the `16 x 16` diagonals stay scalar.
|
||||
- **Neumann/Newton-Schulz inverse (fallback).** `A^{-1} = I - N + N^2 - ... ` is
|
||||
finite (`C` terms) but `O(C)` mma's of `C x C`; Newton-Schulz
|
||||
(`X <- X(2I - AX)`) converges in `~log2(C)` steps for the nilpotent part. Cheap
|
||||
in flops, but more numerically exposed than blocked subst for adversarial decays.
|
||||
Keep as a fallback if blocked subst's register pressure hurts occupancy.
|
||||
|
||||
Verdict: **blocked forward substitution** - it keeps the sensitive diagonal solve
|
||||
exact-in-registers and tensor-core's only the well-conditioned off-diagonal
|
||||
coupling. This is precisely the structure FLA/vLLM use, down-translated to `mma`.
|
||||
|
||||
### Tile/chunk design that fits the 99KB shared budget AND feeds the mma
|
||||
|
||||
The 0031 failure was a layout failure: the all-shared `128x128` f32 state (64KB)
|
||||
crowded out everything and forced `C=16`. The fix is to get the state **out of the
|
||||
bulk shared footprint**. Two complementary mechanisms:
|
||||
|
||||
1. **State register-resident across the chunk loop (the key move).** `S` only
|
||||
participates at chunk boundaries (steps 3,4 at entry; step 6 at exit). Keep it
|
||||
as **mma accumulator fragments distributed across the block's warps** (each
|
||||
warp owns a `dk x dv` sub-tile of `S`), persisting in registers across the
|
||||
sequential chunk loop. Steps 3/4 read `S` as the stationary mma operand; step 6
|
||||
accumulates into it. This **frees the entire 64KB** - shared then holds only the
|
||||
per-chunk K/Q/U/A tiles. (The chunked algorithm's whole point is that the heavy
|
||||
work is intra-chunk and state-free, so the state need not be in shared.)
|
||||
2. **dv-slab tiling for occupancy (the secondary move).** If register pressure
|
||||
from a register-resident `128x128` state caps the kernel at 1 block/SM (likely
|
||||
- that is a lot of accumulator registers), split the `dv=128` value dimension
|
||||
into slabs (`dv_tile in {64, 32}`). Each warp-group owns a `128 x dv_tile`
|
||||
state slab. `A` and the solve depend only on `K` (not `dv`), so they are
|
||||
computed once and the `C x C` `A^{-1}` is **broadcast/recomputed** per slab
|
||||
(cheap once it is mma'd). This shrinks per-block register/shared pressure and is
|
||||
the lever for >1 block/SM.
|
||||
|
||||
**Shared budget at `C = 64` (state register-resident), staging K/Q as bf16/tf32:**
|
||||
|
||||
| Buffer | Elems | Bytes |
|
||||
|---|---|---|
|
||||
| `Kc` (chunk K) | `C x dk = 64x128` | 16KB (bf16) |
|
||||
| `Qc` (chunk Q) | `C x dk` | 16KB (bf16) |
|
||||
| `Uc` (solved U) | `C x dv = 64x128` | 32KB (f32 for the solve) / 16KB (bf16 for the P.U + K^T U mma) |
|
||||
| `A`/`P` scratch | `C x C = 64x64` | 16KB (f32) |
|
||||
| gates `cs/gam/beta` | `~3C` | <1KB |
|
||||
| **state** | (registers) | **0KB shared** |
|
||||
| **Total** | | **~64-80KB** (under the 99KB opt-in) |
|
||||
|
||||
So **`C = 64` fits the 99KB budget once the state is register-resident** - 4x the
|
||||
0031 chunk, and a natural multiple of the `m16n8k*` tiles. For >1 block/SM, drop
|
||||
to `C = 32` + bf16-staged U (`8 + 8 + 16 + 4 = 36KB`, two blocks fit under the
|
||||
~49.5KB/block needed) and/or dv-slab the state. **Recommended default: `C = 64`,
|
||||
tf32 mma, state register-resident** (maximize the BW-saving `C` first; chase the
|
||||
second block/SM only if the bench says occupancy, not BW, is the residual).
|
||||
|
||||
---
|
||||
|
||||
## 3. Occupancy plan (break the 1 block/SM ceiling)
|
||||
|
||||
0031 is pinned to 1 block/SM by the 64KB shared state. The plan, in priority order:
|
||||
|
||||
1. **Free the 64KB: state register-resident** (section 2). This alone may not give
|
||||
2 blocks/SM (the register-distributed `128x128` f32 accumulator is heavy), but
|
||||
it is the precondition for everything and it lets `C` grow to 64 - which is the
|
||||
dominant win (`C`x less state BW). Even at 1 block/SM, `C=64` + mma should flip
|
||||
the sign vs 0031.
|
||||
2. **dv-slab the state** (`dv_tile = 64` then `32`): halve/quarter the per-block
|
||||
accumulator-register and shared pressure to admit a 2nd resident block, at the
|
||||
cost of recomputing the `C x C` `A^{-1}` per slab (cheap on mma). This is the
|
||||
primary occupancy lever once (1) is in.
|
||||
3. **`cp.async` double-buffer the K/Q chunk loads**: overlap the next chunk's
|
||||
global->shared staging with the current chunk's mma, hiding LPDDR5x latency that
|
||||
1-2 blocks/SM cannot. No TMA on sm_121, so plain `cp.async` (`commit_group` /
|
||||
`wait_group`), Ampere-style.
|
||||
4. **Grid starvation at low `n_seqs`** (0031's other failure: grid is `H x n_seqs`,
|
||||
~few hundred blocks): the larger `C` reduces per-block serial chunk steps, and
|
||||
dv-slabbing **multiplies the grid by the slab count** (`H x n_seqs x n_slabs`),
|
||||
directly mitigating the low-`n_seqs` starvation that hurt 0031.
|
||||
|
||||
Honest occupancy caveat: a register-resident `128x128` f32 state is a large
|
||||
register commitment; the realistic outcome is **1-2 blocks/SM**, not high
|
||||
occupancy. The design leans on **mma throughput + cp.async latency hiding + the
|
||||
`C`x BW cut**, not on many resident blocks, to win. If profiling shows the kernel
|
||||
register-capped at 1 block/SM *and* tensor-core-active-% still low, that is the
|
||||
signal to dv-slab harder (smaller `dv_tile`) or accept the achieved win.
|
||||
|
||||
---
|
||||
|
||||
## 4. Bit-exactness + precision risk
|
||||
|
||||
This is a **NEW FP path on top of a NEW FP path**. 0031 is already not byte-equal
|
||||
to the sequential recurrence (different reduction order; README s5 records it as a
|
||||
benign per-path result). Adding tf32/bf16 mma is a *further* reduced-precision
|
||||
step. Gate it exactly like the backend's other new-FP-path precedents
|
||||
(`PAGED_BITEXACT_NOTE.md`, the paged-MoE `8cb0ce23`, the PREFILL_GEMM scope):
|
||||
|
||||
- **Greedy md5 stability** on the standard prompt (README s5 harness) - to catch
|
||||
*unexpected* divergence on the non-prefill paths (decode must stay on the tuned
|
||||
sequential kernel and byte-match its reference; this lever is prefill-only and
|
||||
opt-in, so the default path is untouched).
|
||||
- **`test-backend-ops GATED_DELTA_NET`** at the 0031 prefill shapes (the
|
||||
`S_v=128` exact-multiple / tail / multi-seq / GQA / permuted cases), CUDA0 vs the
|
||||
CPU f32 oracle. **Honest expectation: bf16 mma will likely NOT clear the 1e-7
|
||||
NMSE gate; tf32 is borderline.** So the binding gate is the **KL-gate**, not
|
||||
strict NMSE: require `KLD(tensorcore || f16) <= KLD(sequential || f16)` and PPL
|
||||
within the established band, recorded in `PAGED_BITEXACT_NOTE.md`. tf32 (10-bit
|
||||
mantissa, f32 accumulate) is the precision default precisely to give the KL-gate
|
||||
the best chance.
|
||||
- **Precision fallback ladder if tf32 fails the KL-gate:** (i) **3xtf32**
|
||||
emulation (split each f32 operand into 3 tf32 limbs, 3 mma's, recombine - the
|
||||
CUTLASS fp32-emulation trick; near-f32 accuracy at 3x the mma cost, still far
|
||||
cheaper than serial f32 loops and still a likely net win given the `C`x BW cut);
|
||||
(ii) keep the **decay-coupled and state-boundary products in 3xtf32/f32** while
|
||||
the well-conditioned intra-chunk Gram products use plain tf32 (mixed precision by
|
||||
sensitivity). Do **not** fall back to bf16 for the decay-sensitive terms.
|
||||
- **Preserve the bounded de-gating (section 2):** decays/`gamma`/`beta` stay f32,
|
||||
applied outside the mma. Re-run the adversarial `g in [-20, -1e-4]` op case
|
||||
specifically; a tensor-core kernel that moved a decay inside the mma would be a
|
||||
silent precision regression even if the benign cases pass.
|
||||
|
||||
The likely-favourable framing (as in PREFILL_GEMM): keeping the heavy reductions
|
||||
in f32-accumulate tensor cores is *more* precise than a naive f32 serial loop only
|
||||
if the inputs stay full-width; here inputs are down-cast (tf32/bf16), so this is a
|
||||
genuine precision *trade*, not a free win - hence the KL-gate is mandatory and the
|
||||
3xtf32 ladder exists. Treat NMSE-gate-pass as a bonus, KL-gate-pass as the bar.
|
||||
|
||||
---
|
||||
|
||||
## 5. Honest effort + expected gain
|
||||
|
||||
**This is a multi-week GPU kernel project, not a routing change.** Unlike the
|
||||
PREFILL_GEMM dense lever (a dispatch flip onto an existing vendor kernel), there is
|
||||
no vendor chunked-GDN kernel to route to on sm_121 (CUTLASS/FLA gate the good
|
||||
paths to sm_100a; that is the whole reason vLLM falls back to Marlin on GB10). We
|
||||
must write the `mma` kernel ourselves. Realistic estimate: **4-8 weeks** of
|
||||
focused kernel work, high risk, with non-trivial probability the occupancy/register
|
||||
wall caps the win.
|
||||
|
||||
**Expected gain (mechanism-grounded, section 0/regime-note):** the lever attacks
|
||||
the state-BW that dominates sequential GDN prefill by `~C`x (chunking) while
|
||||
tensor cores absorb the `O(C^2)` intra-chunk flops. Fully realized, it targets
|
||||
vLLM's ~2.5x-cheaper chunked GDN prefill bucket = the ~17% prefill lever the
|
||||
ground-truth attributes to GDN. It should also help the serial-SSM portion of the
|
||||
**decode** residual (README names the irreducible "serial-SSM host loop" as part
|
||||
of the decode floor; a chunked state-update reduces the per-step state traffic
|
||||
there too, though decode `n_tokens` is small so the prefill regime is where it
|
||||
pays). **Honest ceiling:** sm_121 has no wgmma/tcgen05, so we cannot match a
|
||||
hypothetical sm_100a FLA kernel's throughput - the `mma.sync` path is the Ampere-
|
||||
class programming model on Blackwell silicon. But `mma` over serial f32 reductions
|
||||
is an order-of-magnitude flop-rate jump, which is more than enough to flip 0031's
|
||||
-22% into a win and recover most of the GDN prefill bucket. Do not promise full
|
||||
parity with vLLM's sm_100-class kernels; promise "beats the sequential scan and
|
||||
closes most of the GDN prefill gap."
|
||||
|
||||
**Risk register:**
|
||||
- Register-resident `128x128` state may cap occupancy at 1 block/SM (section 3) -
|
||||
mitigated by dv-slabbing, but slabbing recomputes `A^{-1}` per slab.
|
||||
- tf32 may miss the KL-gate -> 3xtf32 ladder (3x mma cost) -> thinner margin.
|
||||
- The win is contingent on prefill GDN being state-BW-bound (regime note); a GPU
|
||||
re-profile that says otherwise kills the lever (step 0).
|
||||
- Blocked-forward-subst register pressure trades against state-register pressure;
|
||||
both compete for the same budget on a 1-block/SM kernel.
|
||||
|
||||
---
|
||||
|
||||
## 6. Phased build plan
|
||||
|
||||
Smallest tensor-core proof-of-concept first, bit-exact/KL-gate + A/B bench at every
|
||||
phase, per `.agents/vllm-parity-methodology.md` (one lever at a time, record
|
||||
rejected/flat variants, ground-truth both engines).
|
||||
|
||||
### Phase 0 - re-confirm the regime on GPU (NO code)
|
||||
nsys a **prefill-only** window (`llama-batched-bench -npp <large> -ntg 0/1`,
|
||||
exclude graph capture) on q36-27b-nvfp4 + q36-35b-a3b, at the backend pin, with
|
||||
`GDN_CHUNK_MIN` set so 0031 runs. Confirm (a) the GDN prefill bucket is
|
||||
state-BW-bound (state memcpy/recurrence dominates, tensor-core-active-% low), and
|
||||
(b) it is ~17% of the prefill step / ~2.5x vLLM's. **If prefill GDN is not
|
||||
state-BW-bound, stop and re-scope** - the entire mechanism (section 0) fails.
|
||||
|
||||
### Phase 1 - PoC: tensor-core just TWO products, same occupancy
|
||||
Keep 0031's `C=16` all-shared layout and 1 block/SM. Replace **only** the two
|
||||
cleanest `C x C` Gram products - step 1 (`KK` for `A`) and step 2 (`QK` for `P`) -
|
||||
with `ggml/src/ggml-cuda/mma.cuh` tf32 tiles (decays still applied in f32 after).
|
||||
Leave the solve, the `S0` products, and the state update serial. This is the
|
||||
minimal "do tensor cores help here at all" probe at fixed occupancy.
|
||||
- Gate: greedy md5 stable; `test-backend-ops GATED_DELTA_NET` prefill shapes via
|
||||
the KL-gate (NMSE if it passes).
|
||||
- Bench: `llama-batched-bench` S_PP, A/B vs sequential and vs 0031-serial, same
|
||||
harness. **If even this does not move S_PP, the head-dim/occupancy is the wall,
|
||||
not the reductions - learn it cheaply before the big build.**
|
||||
|
||||
### Phase 2 - full intra-chunk tensor-core + register-resident state + C=64
|
||||
State register-resident (free the 64KB), `C=64`, tf32 mma for all of steps 1-6,
|
||||
blocked-forward-subst `A^{-1}` (step 7) with mma off-diagonal coupling +
|
||||
in-register `16x16` diagonal solves. Decays/gamma/beta stay f32 throughout.
|
||||
- Gate: as Phase 1, plus the adversarial `g in [-20,-1e-4]` op case explicitly.
|
||||
If tf32 misses the KL-gate, climb the 3xtf32 ladder (section 4).
|
||||
- Bench: S_PP A/B vs sequential, sweep prefill length and `npl`; record the
|
||||
`C in {32,64,128}` sweep and any rejected `C`.
|
||||
|
||||
### Phase 3 - occupancy + latency hiding
|
||||
dv-slab the state (`dv_tile in {64,32}`) for a 2nd resident block and to multiply
|
||||
the grid (fix low-`n_seqs` starvation); `cp.async` double-buffer the K/Q chunk
|
||||
loads. Tune `C`, `dv_tile`, warp count per the bench.
|
||||
- Gate: unchanged (the FP path does not change; this is scheduling).
|
||||
- Bench: final S_PP vs sequential + indicative % of vLLM prefill; name the
|
||||
residual floor honestly (register-cap / sm_121-has-no-tcgen05).
|
||||
|
||||
### Disposition
|
||||
Like 0031, ship **opt-in default-OFF first** (extend the existing `GDN_CHUNK_MIN`
|
||||
gate, add a `GDN_CHUNK_TC` selector if the serial path is kept as fallback). Flip
|
||||
the default only when a separately-built A/B proves S_PP beats the sequential scan
|
||||
*and* the KL-gate holds, recorded in README section 5 + `PAGED_BITEXACT_NOTE.md`.
|
||||
If a phase comes back flat-or-slower, record it as a rejected lever with the reason
|
||||
(the most valuable output if it fails) and keep 0031's serial path as the shipped
|
||||
prefill kernel.
|
||||
|
||||
---
|
||||
|
||||
## 7. Summary
|
||||
|
||||
| Aspect | Decision |
|
||||
|---|---|
|
||||
| Tensor-core ISA | **`mma.sync` only** (sm_121a: no wgmma, no tcgen05/TMEM - confirmed) |
|
||||
| Building block | reuse `ggml/src/ggml-cuda/mma.cuh` tiles, not raw PTX |
|
||||
| Precision default | **tf32** inputs / f32 accumulate; **3xtf32** ladder if KL-gate misses; bf16 only for well-conditioned Gram terms |
|
||||
| Decay handling | gamma/d/beta stay **f32**, applied outside the mma (preserve bounded de-gating) |
|
||||
| A-inverse | blocked forward substitution (FLA UT-transform): in-register diagonal solves + mma off-diagonal |
|
||||
| Chunk size | **C=64** default (4x 0031), C=32 for 2 blocks/SM |
|
||||
| State | **register-resident** (frees the 64KB that forced C=16); dv-slab for occupancy |
|
||||
| Shared budget | ~64-80KB at C=64 state-register-resident (under the 99KB opt-in) |
|
||||
| Mechanism / why it wins | chunking cuts state-BW by ~Cx; mma absorbs the O(C^2) intra-chunk flops the serial 0031 could not |
|
||||
| Bit-exact | NEW per-path; **KL-gate** binding (NMSE likely fails at reduced precision), greedy md5 + adversarial-decay op case |
|
||||
| Effort | **multi-week (4-8 wk), high risk**; no vendor kernel to route to on sm_121 |
|
||||
| Expected gain | beats the sequential scan, closes most of the ~17% GDN prefill bucket toward vLLM's 2.5x; also helps the decode serial-SSM residual. NOT full sm_100-class parity. |
|
||||
| Phasing | P0 re-profile -> P1 two-product PoC -> P2 full intra-chunk + C=64 + reg-state -> P3 occupancy/cp.async; opt-in default-OFF until A/B-proven |
|
||||
|
||||
Decode is untouched (this is prefill-only, opt-in); the stock `llama-cpp` backend
|
||||
stays patch-free. This lever lives entirely in `llama-cpp-localai-paged`.
|
||||
@@ -1,343 +0,0 @@
|
||||
# Layer-2 upstream scope: native fused-GDN kernels for Metal / Vulkan / SYCL
|
||||
|
||||
Source-only analysis (no GPU, no build) of what it would take to give the
|
||||
gated-DeltaNet (GDN / SSM) decode fusions native kernels on the non-CUDA compute
|
||||
backends, so the patch-series decode win extends past CUDA-family hardware.
|
||||
|
||||
This doc is the GDN/SSM-fusion (benefit #1) detail. For the umbrella scope that
|
||||
also covers the paged KV block-table flash-attn read (benefit #2), the free
|
||||
host-side scheduler (benefit #3), the out-of-scope NVFP4 track (benefit #4) and a
|
||||
ROCm note - and the combined per-backend sequencing - see
|
||||
[`ACCELERATOR_PORTING_SCOPE.md`](ACCELERATOR_PORTING_SCOPE.md).
|
||||
|
||||
In our changeset (patches 0018-0030) these fusions ship with CUDA native kernels
|
||||
+ CPU reference kernels ONLY; patch 0030 force-gates them OFF on Metal / Vulkan /
|
||||
SYCL (a CPU-fallback fused op would regress via the device round-trip, and a
|
||||
backend that ran the plain op on the discriminated node would silently
|
||||
miscompute). "Layer 2" is the upstream work that adds the missing native kernels.
|
||||
|
||||
This doc was written against the ggml backend trees in
|
||||
`backend/cpp/llama-cpp-paged-dev` (upstream base #24732, one commit OLDER than the
|
||||
series pin `c299a92c` #25045, with only the two paged-KV patches applied - neither
|
||||
touches GDN/SSM). So every "kernel already exists" statement below is a
|
||||
conservative lower bound: the pin has at least these kernels.
|
||||
|
||||
--------------------------------------------------------------------------------
|
||||
## 0. Headline finding (correct a stale assumption first)
|
||||
|
||||
The series README (section 4c) says "the gated-DeltaNet op has no Vulkan kernel
|
||||
upstream, so the Qwen3.6 hybrid models assert / fall back and don't run there."
|
||||
**That is now stale.** All three backends already carry the BASE compute ops:
|
||||
|
||||
| op | Metal | Vulkan | SYCL |
|
||||
|------------------------|------------------------------------|------------------------------------------|---------------------------------|
|
||||
| GGML_OP_GATED_DELTA_NET| `kernel_gated_delta_net_impl` (f32, NSG 1/2/4) | `gated_delta_net.comp` (d16/32/64/128 x kda, shmem/cluster/nocluster variants) | `gated_delta_net.cpp` (`launch_gated_delta_net<KDA,keep_rs>`) |
|
||||
| GGML_OP_SSM_CONV | `kernel_ssm_conv_f32_f32` (+ `_4`, + batched) | `ssm_conv.comp` (+ APPLY_BIAS, APPLY_SILU specialization consts) | `ssm_conv.cpp` (`kernel_ssm_conv`) |
|
||||
| GGML_OP_SSM_SCAN | yes | `ssm_scan.comp` (mamba2) | `ssm_scan.cpp` (mamba2) |
|
||||
|
||||
Verified: Vulkan `gated_delta_net.comp` was last touched at the upstream base
|
||||
commit (#24732), not by any LocalAI patch. So the GDN COMPUTE op is present on
|
||||
Metal, Vulkan AND SYCL. The Qwen3.6 hybrids therefore DO run on all three today
|
||||
(via the upstream non-fused path that 0030 routes to). The Layer-2 value-add is
|
||||
the decode SPEEDUP from the fusions, NOT enabling the model to run at all.
|
||||
|
||||
Consequence: the GDN-compute op being "partly there" is true on every backend,
|
||||
not just Metal. What is still missing per backend is only the FUSION plumbing
|
||||
(in-place write-back target, the ids gather read, and the conv-update kernel) -
|
||||
a materially smaller scope than "port GDN from scratch."
|
||||
|
||||
--------------------------------------------------------------------------------
|
||||
## 1. Per-op semantics (the four fusions to port)
|
||||
|
||||
All four reuse an existing GGML_OP enum with extra `src[]` slots as a
|
||||
discriminator; none adds a new enum value. f32 throughout. The arithmetic core
|
||||
is IDENTICAL to the upstream non-fused op; only the read source and/or the write
|
||||
target are redirected. That single fact drives the whole bit-exactness story
|
||||
(section 3).
|
||||
|
||||
### OP A - `ggml_gated_delta_net_inplace` (patch 0018)
|
||||
- Enum `GGML_OP_GATED_DELTA_NET`, discriminated by a non-null `src[6]` =
|
||||
`state_dst` (a contiguous `[S_v*S_v*H, n_seqs]` view into the recurrent-state
|
||||
cache at `kv_head`). K == 1 only.
|
||||
- Semantics: run the standard GDN recurrence, but write the FINAL recurrent state
|
||||
directly into `state_dst` instead of appending it to the op output. The op
|
||||
output then carries only the attention scores. Removes the per-layer per-step
|
||||
~full-state D2D copy-back (the 0018 win).
|
||||
- Race (in-place read == write): each (seq, head) block owns a disjoint cache
|
||||
slot. The kernel loads the whole prior state `s0` into per-thread registers
|
||||
(`s_shard` on CUDA, `ls[NSG]` on Metal, the column shard on Vulkan/SYCL)
|
||||
BEFORE the ring write, so reading and writing the same slot is safe.
|
||||
|
||||
### OP B - `ggml_gated_delta_net_inplace_ids` (patch 0019)
|
||||
- Adds `src[5]` = FULL state cache `[S_v,S_v,H,n_rs_slots]`, `src[7]` = `ids`
|
||||
(I32, per-seq source slot == the recurrent-state `s_copy`), `op_param[1]` =
|
||||
`rs_head` (destination base slot). Still has the OP-A `src[6]` in-place target.
|
||||
- Semantics: read each sequence's prior state directly from `cache[ids[seq]]`
|
||||
(mirrors `ggml_ssm_scan`'s ids source), eliminating the `ggml_get_rows`
|
||||
materialization. Combined with OP A the op now reads AND writes the cache in
|
||||
place.
|
||||
- Race: identity sequences (`ids[s] == rs_head + s`, the steady AR-decode case)
|
||||
read s0 in place from the destination slot (safe via the register snapshot
|
||||
above). Non-identity sequences (reorder / rs_zero remap) are first copied by a
|
||||
TINY separate gather kernel (`gdn_gather_nonident`, one block/seq) into a
|
||||
DISJOINT scratch that the recurrence then reads, so the recurrence never reads
|
||||
a slot another block is writing. Value-preserving memcpy -> bit-identical to
|
||||
the get_rows path.
|
||||
|
||||
### OP C - `ggml_ssm_conv_update_inplace` (patch 0021)
|
||||
- Enum `GGML_OP_SSM_CONV`, discriminated by a non-null `src[3]` =
|
||||
`conv_state_dst` (`[(K-1)*channels, n_seqs]` in-place ring view).
|
||||
`src[0]` = conv_states `[K-1, channels, n_seqs]`, `src[1]` = conv_kernel
|
||||
`[K, channels]`, `src[2]` = x_cur `[channels, 1, n_seqs]`. `op_param[0]` =
|
||||
fuse_silu.
|
||||
- Semantics (decode, n_seq_tokens == 1): per (channel, sequence) assemble the
|
||||
width-K conv window in registers from the K-1 cached taps + the current token,
|
||||
compute the depthwise conv with the SAME ascending-tap FMA order as plain
|
||||
`ssm_conv` (`tap0*w0 + ... + xc*w_{K-1}`, then `+0.0f` to match plain conv's
|
||||
`sumf += b` with b==0), optionally fold SiLU, write the conv output
|
||||
`[channels,1,n_seqs]`, and write the 1-token-shifted ring state back in place.
|
||||
Replaces the 4-op decode conv chain (transpose + concat + conv + silu + ring
|
||||
cpy).
|
||||
- Race: read source (gathered taps) and write target (cache view) are disjoint
|
||||
buffers -> race-free by construction, no ids/identity logic.
|
||||
|
||||
### OP D - `ggml_ssm_conv_update_inplace_ids` (patch 0028)
|
||||
- Same enum, discriminated by a non-null `src[4]` = `ids`; `src[0]` becomes the
|
||||
FULL conv cache `[K-1, channels, n_cells]`; `op_param[1]` = rs_head.
|
||||
- Semantics: gather-free conv-update - read each sequence's prior taps from
|
||||
`cache[ids[s]]` in-kernel (no get_rows). Identity reads in place from
|
||||
`conv_state_dst`; non-identity gathered into a disjoint scratch first by a tiny
|
||||
`ssm_conv_gather_nonident` kernel. The window is copied to a local array
|
||||
BEFORE the (possibly aliasing) ring write so the identity read==write slot is
|
||||
correct. Bit-identical to get_rows + OP C.
|
||||
|
||||
### Net new kernels vs reuse, per op
|
||||
- OP A: NOT a new compute kernel - a write-target redirection of the EXISTING
|
||||
GDN kernel + 1 buffer binding + a supports_op/op-handler branch.
|
||||
- OP B: the GDN kernel gains a per-seq read-base select (identity vs scratch) +
|
||||
1 ids binding + rs_head param + 1 tiny gather kernel.
|
||||
- OP C: a GENUINELY NEW kernel on each backend. The existing `ssm_conv` computes
|
||||
a windowed reduction over a PRE-concatenated input; it does not assemble the
|
||||
window from cached taps + the current token, fold silu, or write the shifted
|
||||
ring state. This is the largest net-new piece.
|
||||
- OP D: the OP-C kernel gains the read-base select + 1 ids binding + rs_head + 1
|
||||
tiny conv gather kernel.
|
||||
|
||||
The `ggml.h` / `ggml.c` builders, the CPU reference kernels, the model-graph
|
||||
emission (`delta-net-base.cpp`, qwen35*), and the `test-backend-ops` cases are
|
||||
SHARED and already done by patches 0018/0019/0021/0028. The only NEW per-backend
|
||||
work is the kernel(s) + the backend wiring.
|
||||
|
||||
--------------------------------------------------------------------------------
|
||||
## 2. Per-backend: authoring model, effort, gotchas, wiring
|
||||
|
||||
### 2.1 Metal (MSL)
|
||||
|
||||
Authoring model: `.metal` MSL source (`ggml-metal.metal`), function-constant
|
||||
specialization (e.g. `FC_GATED_DELTA_NET`), kernels templated on `NSG`; host
|
||||
glue split across `ggml-metal-ops.cpp` (`ggml_metal_op_*` encode), the pipeline
|
||||
lookup in `ggml-metal-device.cpp`/`.m`, the kargs struct in `ggml-metal-impl.h`,
|
||||
and `supports_op` in `ggml-metal-device.m`. Threadgroup model; Apple GPU
|
||||
simdgroup width is a FIXED 32, `simd_sum` for the per-column reduce.
|
||||
|
||||
Effort: MEDIUM. ~350-500 LOC. The GDN and plain-ssm_conv kernels already exist
|
||||
and are ergonomic to extend. OP A is a write-base redirect of the existing
|
||||
`kernel_gated_delta_net_impl` (its tail already does
|
||||
`dst_state = dst + attn_size + state_out_base; dst_state[is] = ls[j]` after
|
||||
loading `ls[]` into registers - just point `dst_state` at the `state_dst` buffer
|
||||
and add the binding). OP C is the one net-new MSL kernel (Metal has NO bias/silu
|
||||
ssm_conv variant today - only plain + `_4` + batched - so the silu-fold and ring
|
||||
write are both new). Host glue spans 3-4 files.
|
||||
|
||||
Gotchas:
|
||||
- In-place race: the existing kernel ALREADY snapshots the state column into
|
||||
`ls[NSG]` registers before writing, so OP A/B are safe with no barrier; OP C/D
|
||||
must mirror the `float window[K]` local-copy-before-write that CPU/CUDA use.
|
||||
- Discriminated SSM_CONV: `supports_op` for `GGML_OP_SSM_CONV` currently returns
|
||||
`has_simdgroup_reduction` with NO check of `src[3]`/`src[4]`; GDN returns
|
||||
`has_simdgroup_reduction && src[2]->ne[0] % 32 == 0` with NO check of
|
||||
`src[6]`/`src[7]`. Both must be tightened (accept the discriminated variant
|
||||
only once the kernel exists) AND `ggml_metal_op_ssm_conv` /
|
||||
`ggml_metal_op_gated_delta_net` must branch on the extra src to pick the kernel.
|
||||
- Bit-exactness: fixed 32-wide simdgroup makes this the SIMPLEST of the three -
|
||||
the fused variant only redirects addresses, so it is bit-identical to Metal's
|
||||
own non-fused path by construction (the conv per-channel FMA needs the exact
|
||||
ascending order + the `+0.0f`).
|
||||
- The kargs struct grows by the `state_dst` / `ids` / `rs_head` fields; a new
|
||||
pipeline name (or a function-constant branch) distinguishes the variants.
|
||||
|
||||
### 2.2 Vulkan (GLSL .comp -> SPIR-V)
|
||||
|
||||
Authoring model: GLSL `.comp` in `vulkan-shaders/`, compiled at build time by
|
||||
`vulkan-shaders-gen` into embedded SPIR-V byte arrays (`gated_delta_net_f32_data`
|
||||
etc.); pipeline creation in `ggml-vulkan.cpp` declares the binding count +
|
||||
push-constant size; a push-constant struct per op; host dispatch `ggml_vk_*`
|
||||
binds subbuffers; `supports_op` in the device support function. Subgroup size
|
||||
VARIES by vendor (NVIDIA 32, AMD 64, Intel 8/16/32).
|
||||
|
||||
Effort: HARDEST. ~450-650 LOC + the most build/host glue. Same kernel logic as
|
||||
Metal/SYCL, but every new shader or variant requires: the shaders-gen regen, a
|
||||
new `ggml_vk_create_pipeline` registration with an explicit binding count and
|
||||
push-constant size, a new/extended push-constant struct (add `rs_head`), and
|
||||
GROWING the descriptor binding set from the current 7 (`src[0..5]` + dst) to 8-9
|
||||
(`state_dst`, `ids`). The GDN host dispatch hardcodes a 6-src bind loop and the
|
||||
pipeline is created with `"main", 7, ...` - both must change.
|
||||
|
||||
Gotchas:
|
||||
- Subgroup variance interacts with the EXISTING variant matrix: the GDN comp
|
||||
already ships shmem / cluster / nocluster variants keyed on subgroup size and
|
||||
relies on `S_V % COLS_PER_WG == 0`. The OP-A/B read/write redirect must be
|
||||
applied across ALL of those variants, and re-validated per vendor.
|
||||
- In-place race: GLSL must read the full column shard into local registers before
|
||||
the ring write (same pattern); confirm the SPIR-V memory model is not relied on
|
||||
for cross-invocation ordering (it is not - blocks are disjoint per (seq,head)).
|
||||
OP C/D need the explicit window-to-local copy.
|
||||
- Discriminated SSM_CONV: `supports_op` returns `op->src[0]->type == F32` with NO
|
||||
discriminator check; GDN loops `src[0..5]` F32 with NO `src[6]`/`src[7]` check.
|
||||
Both must be tightened. This is the backend where the 0030 hazard is most
|
||||
concrete (a present plain-conv kernel + a permissive supports_op = silent
|
||||
miscompute) - Vulkan is the exact case 0030 was written for.
|
||||
- conv-update is per-channel (one invocation per channel) so it is
|
||||
subgroup-AGNOSTIC; only the GDN recurrence carries the subgroup-width burden.
|
||||
- Vulkan's `ssm_conv.comp` ALREADY has APPLY_SILU + APPLY_BIAS specialization
|
||||
constants, so the silu-fold half of OP C is partly precedented here (unlike
|
||||
Metal); the ring write-back + tap-window assembly are still new.
|
||||
|
||||
### 2.3 SYCL (single-source DPC++)
|
||||
|
||||
Authoring model: plain C++ `.cpp`/`.hpp` per op (`gated_delta_net.cpp`,
|
||||
`ssm_conv.cpp`); a SYCL `queue.parallel_for` over an `nd_range` with
|
||||
`reqd_sub_group_size(WARP_SIZE)`; sub-group reductions (`warp_reduce_sum`);
|
||||
`supports_op` in `ggml-sycl.cpp`. NO separate shader-compile step (single
|
||||
source).
|
||||
|
||||
Effort: EASIEST to author. ~250-350 LOC. The SYCL op handlers + kernels are
|
||||
near-VERBATIM mirrors of the CUDA ones (`launch_gated_delta_net<KDA,keep_rs>`,
|
||||
`s_shard`, `curr_state`, `state = dst + attn_score_elems`, `warp_reduce_sum`) -
|
||||
a dpct/SYCLomatic-style port. The CUDA diffs in 0018/0019/0021/0028 would port
|
||||
almost line-for-line: add the `state_dst` param, the `ids`/`rs_head` params, the
|
||||
read-base select, the two tiny gather kernels, and the new conv-update kernel.
|
||||
No pipeline/push-constant/binding bookkeeping.
|
||||
|
||||
Gotchas:
|
||||
- In-place race: the `s_shard[]` / window arrays are per-work-item private, so
|
||||
the register-snapshot-before-write pattern carries over directly. Safe.
|
||||
- Discriminated SSM_CONV: `supports_op` checks `src[0]`/`src[1]` F32 with NO
|
||||
discriminator check; GDN returns a BARE `true` (the MOST permissive, so the
|
||||
hazard is worst here). Both must be tightened, and `ggml_sycl_op_ssm_conv` /
|
||||
`ggml_sycl_op_gated_delta_net` must branch on the extra src.
|
||||
- Bit-exactness: `WARP_SIZE` is compile-fixed (Intel sub-group 8/16/32), same
|
||||
situation as CUDA; the fused variant matches SYCL's own non-fused path by
|
||||
construction. conv-update is per-channel -> subgroup-agnostic.
|
||||
|
||||
### 2.4 Common wiring (all three) + the 0030 emission-gate change
|
||||
|
||||
Per backend, four wiring touch-points beyond the kernel body:
|
||||
1. `supports_op`: tighten the `GGML_OP_SSM_CONV` and `GGML_OP_GATED_DELTA_NET`
|
||||
entries so the discriminated/extra-src node is reported supported ONLY when
|
||||
the new kernel handles it (and rejected otherwise, instead of today's
|
||||
silently-true-for-the-plain-kernel).
|
||||
2. op handler: branch on `src[3]`/`src[4]` (conv) and `src[6]`/`src[7]` (GDN) to
|
||||
dispatch the fused kernel.
|
||||
3. pipeline/kernel registration (Vulkan: + push-constant struct + descriptor
|
||||
bindings; Metal: + kargs fields + pipeline name; SYCL: just the new functions).
|
||||
4. The patch-0030 gate in `src/llama-context.cpp`.
|
||||
|
||||
The 0030 change today is a hard allow-list: any non-CPU compute backend whose reg
|
||||
name is not `"CUDA"`/`"ROCm"`/`"MUSA"` forces `fused_gdn_ar = fused_gdn_ch =
|
||||
auto_fgdn = false`. As each backend gains kernels this must become capability-
|
||||
driven, in one of two ways:
|
||||
- minimal: add the backend's reg name (e.g. `"Metal"`) to the allow-list once its
|
||||
kernels + tightened supports_op ship; OR
|
||||
- clean (recommended upstream form): DELETE the name allow-list and make
|
||||
`supports_op` authoritative - have the `auto_fgdn` resolution probe
|
||||
`ggml_backend_dev_supports_op` on a representative node that carries the
|
||||
discriminated `src[]` slots. Then routing falls out of the normal scheduler
|
||||
fallback and no backend name is ever hard-coded. This also fixes 0030's stated
|
||||
weakness that the upstream `auto_fgdn` check only inspects GATED_DELTA_NET
|
||||
nodes and covered the discriminated SSM_CONV only incidentally.
|
||||
|
||||
--------------------------------------------------------------------------------
|
||||
## 3. Bit-exactness per backend (the md5 gate question)
|
||||
|
||||
Feasible on ALL THREE, and not actually constraining, because of how the gate is
|
||||
scoped:
|
||||
|
||||
- The series md5 gate is a CUDA-vs-CPU comparison; each GPU backend ALREADY has
|
||||
its own f32 reduction order (Metal `simd_sum`, Vulkan subgroup reduce, SYCL
|
||||
`warp_reduce_sum`) that differs from CUDA's and from CPU's. There is no
|
||||
cross-backend md5 and none is expected.
|
||||
- The relevant per-backend invariant is: the FUSED variant must equal that
|
||||
backend's OWN non-fused path. The fusions change only the read source
|
||||
(gather -> indexed read; the gather is a value-preserving memcpy) and the write
|
||||
target (appended output -> in-place cache slot). They do NOT touch the
|
||||
per-column FMA/reduce order. So the fused op is bit-identical to the
|
||||
non-fused op on the same backend BY CONSTRUCTION.
|
||||
- Two arithmetic details each port MUST preserve exactly: (a) the conv
|
||||
ascending-tap order plus the `+0.0f` that matches plain `ssm_conv`'s
|
||||
`sumf += b` with b==0; (b) the existing GDN per-column subgroup reduce (do not
|
||||
re-order it). Get those right and `test-backend-ops` (backendX-vs-CPU, already
|
||||
registered for SSM_CONV / SSM_CONV_UPDATE / SSM_CONV_UPDATE_IDS /
|
||||
GATED_DELTA_NET) is the per-backend gate.
|
||||
|
||||
--------------------------------------------------------------------------------
|
||||
## 4. Upstream path and ranked recommendation
|
||||
|
||||
### Ops-first, then one PR per backend (NOT one big PR)
|
||||
|
||||
Recommended sequence:
|
||||
|
||||
1. PR #1 - OPS (already essentially done, upstreamable as-is): the `ggml.h`/
|
||||
`ggml.c` builders, the CPU reference kernels, the CUDA kernels, the
|
||||
`test-backend-ops` cases, and the capability-driven gate (the clean
|
||||
`supports_op`-authoritative version of 0030). This is independently mergeable
|
||||
and mirrors how llama.cpp lands new ops (CPU + CUDA first; GDN itself landed
|
||||
that way).
|
||||
2. PR #2 - Metal kernels + wiring.
|
||||
3. PR #3 - SYCL kernels + wiring.
|
||||
4. PR #4 - Vulkan kernels + wiring.
|
||||
|
||||
Do NOT bundle the backends: each needs its own hardware to validate
|
||||
`test-backend-ops`, reviewers are backend-specialized, and a regression in one
|
||||
must not block the others.
|
||||
|
||||
### Value x effort ranking (which backend first)
|
||||
|
||||
| backend | user base / value | author effort | bit-exact difficulty | net rank |
|
||||
|---------|----------------------------|---------------|----------------------|----------|
|
||||
| Metal | HIGH (Apple Silicon = largest non-CUDA LocalAI base; unified memory makes the no-copy / no-gather plumbing wins map directly) | MEDIUM | LOWEST (fixed 32 simdgroup) | **1st** |
|
||||
| SYCL | LOW-MED (Intel GPU) | LOWEST (near-verbatim CUDA mirror) | LOW | **2nd** |
|
||||
| Vulkan | HIGHEST breadth (AMD + Intel + cross-vendor) | HIGHEST (shaders-gen + variant matrix + subgroup variance + descriptor growth) | MEDIUM (per-vendor subgroup validation) | **3rd** |
|
||||
|
||||
Recommendation: **Metal first.** It banks the biggest user-facing decode win at
|
||||
medium effort, the base GDN + conv kernels already exist, and Apple's fixed
|
||||
simdgroup width makes bit-exactness the simplest. **SYCL second** as a cheap,
|
||||
nearly mechanical follow-on (the port is a line-for-line CUDA mirror, so it is
|
||||
low-cost insurance even though the Intel-GPU audience is smaller). **Vulkan last**
|
||||
as the high-effort / high-breadth capstone - it reaches the widest hardware
|
||||
(AMD + Intel + anything with a Vulkan driver), but the shader-gen pipeline, the
|
||||
existing variant matrix, the subgroup-width variance, and the per-vendor
|
||||
validation burden make it the right capstone once the pattern is proven on
|
||||
Metal + SYCL.
|
||||
|
||||
A reasonable cheaper variant: ship Metal + SYCL together right after the ops PR
|
||||
(both are register-snapshot ports with no shader-gen step) and treat Vulkan as a
|
||||
separate later effort.
|
||||
|
||||
--------------------------------------------------------------------------------
|
||||
## 5. Summary
|
||||
|
||||
- GDN-compute and plain SSM_CONV kernels ALREADY EXIST on Metal, Vulkan and SYCL
|
||||
(the README's "no Vulkan kernel" line is stale). The Qwen3.6 hybrids run on all
|
||||
three today via the non-fused path; Layer-2 is about the decode SPEEDUP.
|
||||
- Per backend the NEW work is: redirect the GDN state write (OP A) + add the ids
|
||||
read (OP B) to the existing GDN kernel, write ONE new conv-update kernel
|
||||
(OP C) + its ids variant (OP D), add two tiny gather kernels, and tighten
|
||||
supports_op + the op-handler branch + (Vulkan) the pipeline/push-constant/
|
||||
descriptor wiring. The builders, CPU refs, model graph and tests are shared and
|
||||
already done.
|
||||
- Bit-exactness is feasible everywhere and per-backend by construction (the
|
||||
fusions redirect addresses, not the f32 reduction order); `test-backend-ops`
|
||||
(backendX-vs-CPU) is the gate.
|
||||
- Sequence: ops-first PR (incl. the capability-driven replacement for 0030's
|
||||
name allow-list), then Metal, then SYCL, then Vulkan.
|
||||
@@ -1,417 +0,0 @@
|
||||
# vLLM Parity - Final State (Qwen3.6 NVFP4 on GB10)
|
||||
|
||||
> 2026-06-30 update: this document records the earlier final-state verdict. The
|
||||
> investigation has since been reopened; see `GB10_PARITY_REOPEN_SPEC.md`,
|
||||
> `GB10_PARITY_PHASE0_RESULTS.md`, and the active `docs/superpowers/plans/`
|
||||
> Phase 6/Phase 7 files for the current measured state and follow-up scope.
|
||||
|
||||
> **Status: CLOSED.** This is the standing record of the exhaustive GB10 (DGX
|
||||
> Spark, sm_121) parity investigation for `llama-cpp-localai-paged` against vLLM
|
||||
> on the Qwen3.6 hybrid gated-DeltaNet NVFP4 models. It exists so the
|
||||
> investigation is **never re-litigated**: every lever attempted, its verdict,
|
||||
> its key number, and the structural floors that bound the result are recorded
|
||||
> below with the artifact each number came from. The one-line conclusion:
|
||||
> **prefill is genuinely capped at 36-43% of vLLM (FP4-MMQ optimality + GDN
|
||||
> O(C^2) intra-chunk complexity; prefill is not CUDA-graph-replayed, so these are
|
||||
> real floors, not profiling artifacts); decode-serving is near-parity at ~86% of
|
||||
> vLLM's true GPU-steady decode (the long-standing ~56% headline was a
|
||||
> measurement / operating-point artifact, corrected below), with the residual
|
||||
> ~14% being vLLM's mature fused-Marlin + Triton-elementwise kernels that are not
|
||||
> cheaply replicable on GB10.**
|
||||
|
||||
Companion docs (design/rationale, not re-summarized here): the patch-series
|
||||
[`README.md`](../README.md) (section 5 dev-notes), `VLLM_PARITY_LEVER_MAP.md`,
|
||||
`PREFILL_GEMM_SCOPE.md`, `PREFILL_GEMM_RESULTS.md`, `DECODE_SERVING_SCOPE.md`,
|
||||
`TENSORCORE_GDN_SCOPE.md`, `TENSORCORE_GDN_BUILD_PLAN.md`, `PAGED_BITEXACT_NOTE.md`.
|
||||
|
||||
Source key (every number below cites one of these):
|
||||
- **CDEF** = the definitive same-session both-engine run `dgx:~/bench/COMBINED_DEFINITIVE.txt` (2026-06-29, GIT_HEAD `a7d439e`, h2h_cli3 OpenAI `/v1/completions`, fresh-nonce prompts, ignore_eos, ptok128 gen128; paged `LLAMA_KV_PAGED=1 LLAMA_MOE_FORCE_GRAPHS=1`, GDN M5 on, S1 on, S3 off; vLLM 0.23.0 gpu-util 0.85 max-model-len 4096 max-num-seqs 256 tp1).
|
||||
- **README** = the static `llama-batched-bench` table in [`README.md`](../README.md) section 4 (npp128/ntg128; patched vs stock-`9d5d882d` vs vLLM-prior).
|
||||
- **PGR** = `PREFILL_GEMM_RESULTS.md`. **LMAP** = `VLLM_PARITY_LEVER_MAP.md` (profile-validated section). **DSS** = `DECODE_SERVING_SCOPE.md`. **MG** = `dgx:~/bench/marlin_gate/`. **GDNAB** = `dgx:~/bench/gdn_p1_ab/`. **0034/0035** = patch headers in `patches/paged/`.
|
||||
- **HNP** = the clean, uncontended, **graph-node-traced** both-engine high-N decode profile (2026-06-30): `dgx:~/highN_prof2/*.nsys-rep` (paged, npl=256) + `dgx:~/highN_vllm/*.nsys-rep` (vLLM), captured with `nsys --cuda-graph-trace=node` and decomposed by the **difference method** (per-token cost = ntg=64 profile minus ntg=16 profile). **This supersedes every earlier decode decomposition** (LMAP included): those were taken without `--cuda-graph-trace=node`, which collapses each graph replay into one opaque launch and made the per-kernel decode attribution an artifact (see 2c).
|
||||
- "estimated" marks any figure not pinned to one of the above.
|
||||
|
||||
---
|
||||
|
||||
## 1. The benchmark (paged vs vLLM vs stock)
|
||||
|
||||
Two models: the MoE **Qwen3.6-35B-A3B-NVFP4** (decision model, 256 experts top-8,
|
||||
30 GDN + 10 full-attn layers + a dense shared expert per layer) and the dense
|
||||
**Qwen3.6-27B-NVFP4** (48 GDN + 16 full-attn). All numbers GB10 / CUDA 13 /
|
||||
sm_121. The current backend pin is `0ed235ea2c17a19fc8238668653946721ed136fd`;
|
||||
the CDEF benchmark artifact itself records the dev-tree commit that produced
|
||||
those binaries.
|
||||
|
||||
### 1a. Prefill (S_PP, prefill tokens/s)
|
||||
|
||||
Paged = static `llama-batched-bench` PP block; vLLM = server prefill-phase rate
|
||||
at the same prompt length. Source: **CDEF**.
|
||||
|
||||
| Model | shape | paged S_PP | vLLM S_PP | paged % of vLLM |
|
||||
|---|---|---:|---:|---:|
|
||||
| MoE 35B-A3B | PP=512, B=32 | 2309.6 | 6418.9 | **36.0%** |
|
||||
| MoE 35B-A3B | PP=2048, B=32 | 2401.9 | 6748.5 | **35.6%** |
|
||||
| Dense 27B | PP=512, B=32 | 960.3 | 2277.3 | **42.2%** |
|
||||
| Dense 27B | PP=2048, B=32 | 1010.2 | 2360.1 | **42.8%** |
|
||||
|
||||
Prefill is the largest absolute gap. The profile-validated decomposition (LMAP,
|
||||
nsys both-engine, MoE decision model) attributes it as: paged **395.9 us/tok** vs
|
||||
vLLM **197.0 us/tok** (total gap ~198.9 us/tok), split GDN **+59.2** (~30%),
|
||||
MoE-GEMM **+56.5** (~28%), ew/layout/glue **+21.4** (~11%), act-quant **+15.2**
|
||||
(~8%), bf16-proj **+13.7** (~7%), gate **+12.4** (~6%), norms **+11.1** (~6%),
|
||||
dispatch **+5.9** (~3%).
|
||||
|
||||
### 1b. Decode / serving (per-seq + aggregate decode t/s), staggered serving
|
||||
|
||||
Source: **CDEF** NPL runs (continuous serving via h2h_cli3). `decode_agg` =
|
||||
aggregate decode t/s; `perseq` = decode tok/s/seq; PEAK_GB = peak process VRAM.
|
||||
|
||||
**MoE Qwen3.6-35B-A3B-NVFP4:**
|
||||
|
||||
| N | paged decode_agg | vLLM decode_agg | paged perseq | vLLM perseq | perseq % of vLLM | paged TTFT_mean ms | vLLM TTFT_mean ms | paged PEAK_GB | vLLM PEAK_GB |
|
||||
|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|
|
||||
| 8 | 208.1 | 297.1 | 25.68 | 36.68 | **70.0%** | 747.9 | 204.2 | 50.03 | 112.42 |
|
||||
| 32 | 379.1 | 575.7 | 11.40 | 17.49 | **65.2%** | 2377.9 | 640.8 | 52.13 | 112.20 |
|
||||
| 128 | 611.9 | 958.2 | 4.14 | 6.97 | **59.4%** | 7058.3 | 1965.4 | 60.57 | 112.51 |
|
||||
| 256 | 717.8 | 1177.4| 2.29 | 4.12 | **55.6%** | 13533.6 | 3937.3 | 70.18 | 112.55 |
|
||||
|
||||
**Dense Qwen3.6-27B-NVFP4:**
|
||||
|
||||
| N | paged decode_agg | vLLM decode_agg | paged perseq | vLLM perseq | perseq % of vLLM | paged TTFT_mean ms | vLLM TTFT_mean ms | paged PEAK_GB | vLLM PEAK_GB |
|
||||
|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|
|
||||
| 8 | 84.0 | 72.1 | 10.42 | 8.93 | **116.7%** | 1914.7 | 493.1 | 77.97 | 109.63 |
|
||||
| 32 | 196.5 | 214.7 | 5.83 | 6.56 | **88.9%** | 7023.3 | 1735.4 | 83.04 | 109.65 |
|
||||
| 128 | 343.8 | 431.8 | 2.18 | 3.10 | **70.3%** | 19468.9 | 5455.0 | 101.93 | 109.67 |
|
||||
| 256 | 380.3 | 532.5 | 1.13 | 1.82 | **62.1%** | 36306.8 | 10824.1 | 114.63 | 109.67 |
|
||||
|
||||
End-to-end aggregate `agg_tps` (incl. prefill contention), **CDEF**: MoE paged
|
||||
179.7/301.4/425.6/459.9 vs vLLM 278.5/515.6/798.3/915.4 at N=8/32/128/256; dense
|
||||
paged 72.6/141.4/205.8/213.3 vs vLLM 69.4/193.3/346.6/394.7.
|
||||
|
||||
**Reading the table.** Dense decode is **ahead of vLLM at low concurrency
|
||||
(116.7% at N=8)**. The high-N percentages here (perseq ~56%, decode_agg ~61% at
|
||||
N=256) are **server-window** numbers and **understate true engine parity**: they
|
||||
divide the paged serving rate by vLLM's *prefill-overlap-inflated* server rate.
|
||||
The corrected, graph-node-traced decomposition (section 2c, **HNP**) shows paged
|
||||
decode at **~86% of vLLM's true GPU-steady decode**, with the remaining
|
||||
server-window gap being an S3-recoverable serving graph-reuse overhead (2d). The
|
||||
earlier "this is just the bandwidth floor / vLLM pays equally" reading was a
|
||||
**profiling artifact** and is corrected in 2c.
|
||||
|
||||
**PEAK_GB is the structural memory advantage.** vLLM's PEAK_GB is a **fixed
|
||||
~109-112.5 GB reservation** (the `--gpu-memory-utilization 0.85` block-manager
|
||||
pre-allocation of the ~128 GB unified LPDDR5x) and does **not** vary with N. The
|
||||
paged backend allocates KV on demand, so its peak **grows with load** but stays
|
||||
far below vLLM at low/mid concurrency: MoE N=8 uses **50.0 vs 112.4 GB (~2.2x
|
||||
less)**, and even at N=256 MoE is 70.2 vs 112.6 GB. This is the headline of
|
||||
section 5 (memory advantage / higher max concurrency per GPU) and is real,
|
||||
bit-exact, and not an operating-point trick.
|
||||
|
||||
### 1c. Patched vs true-stock (static batched-bench, the patch-series multiplier)
|
||||
|
||||
Stock `9d5d882d` was not in the same-session CDEF run; the patched-vs-stock
|
||||
multiplier is the static `llama-batched-bench` table (**README**, npp128/ntg128,
|
||||
decode t/s):
|
||||
|
||||
| | N=8 | N=32 | N=64 | N=128 | max x over stock |
|
||||
|---|---:|---:|---:|---:|---:|
|
||||
| Dense patched / stock | 85.3 / 68.3 | 211.9 / 119.9 | 305.2 / 142.8 | 382.1 / 155.1 | **2.46x** |
|
||||
| MoE patched / stock | 230.3 / 186.7 | 466.4 / 267.4 | 622.4 / 320.5 | 784.3 / 347.2 | **2.26x** |
|
||||
|
||||
In that **static** regime the patched decode kernel is **at vLLM parity**
|
||||
(dense 121/100/99/91% of vLLM-prior across widths; MoE 90/93/91/89%). The serving
|
||||
table in 1b is the harder continuous regime; the gap between the two regimes is
|
||||
the subject of section 2 (serving) and was fully closed on the host side.
|
||||
|
||||
---
|
||||
|
||||
## 2. Complete lever map (every attempt, verdict, key number)
|
||||
|
||||
Bit-exactness convention (per `PAGED_BITEXACT_NOTE.md`): the gate is **per-path**.
|
||||
Dense greedy md5 `5951a5b4`; paged-MoE greedy md5 `8cb0ce23` (a benign
|
||||
FP-accumulation-order reorder vs non-paged `07db32c2`, KL-validated). "BE" = greedy
|
||||
md5 byte-identical; "KL-benign" = new FP path, gated by KL-divergence within band.
|
||||
|
||||
### 2a. PREFILL - weight GEMM track (verdict: FP4-MMQ is optimal on GB10)
|
||||
|
||||
Four kernels were built or ported to beat MMQ at large-M MoE prefill. **All
|
||||
rejected; FP4-MMQ stays the shipped path.** The decisive surprise (LMAP, both-engine
|
||||
nsys): **on sm_121 vLLM itself does not run native FP4** - it runs **Marlin W4A16**
|
||||
(FP4 dequant to bf16 in-register + bf16 GEMM) for experts and FP8 projections,
|
||||
capped at bf16-tensor-core peak (~half FP4 peak). So MMQ's native FP4 path is
|
||||
already structurally competitive on this exact silicon.
|
||||
|
||||
| Lever | What | Verdict | Key number | Source |
|
||||
|---|---|---|---|---|
|
||||
| **0033** dequant -> bf16 cuBLAS | route large-M NVFP4 dense GEMM off MMQ to dequant->bf16 nvjet/cuBLAS | **REJECTED** (regression) | dense S_PP **-49% / -42% / -29%** at M=512/1024/2048; bit-exact md5 identical, KL-better | PGR |
|
||||
| dense-cuBLAS reroute (full sweep) | the same reroute across the dense + MoE prefill sweep | **REJECTED** | **-31% to -62%** band (estimated; the artifact-pinned dense subset is -29% to -49%, PGR) | LMAP / recorded verdict |
|
||||
| **0034** native FP4-MMA W4A4 | Blackwell `mxf4nvf4` OMMA large-M kernel, PoC verbatim | **REJECTED in-backend** | PoC `~103 TFLOP/s` (57.7% of FP4 peak, beats cuBLAS-bf16, NMSE=0), but the standalone PoC win **did not hold in-backend** | 0034 header / LMAP |
|
||||
| **0035** W4A16-Marlin grouped MoE | FP4->bf16 in-register dequant + bf16 `mma.sync`, zero act-quant tax (vLLM's exact sm_121 shape) | **REJECTED** (perf regression) | correct + bit-exact-gated: `test-backend-ops MUL_MAT_ID` 81/81; KL **benign and better** (marlin KLD **0.131** < MMQ **0.137**, same-top-p 84.6% vs 84.3%); md5 short identical, long one benign flip - but **-39%** S_PP vs MMQ (estimated/recorded; MG holds only the correctness+KL gate) | 0035 header, MG |
|
||||
| offline-repack Marlin / vLLM-verbatim Marlin | repack weights offline to Marlin layout; port vLLM's Marlin kernel verbatim | **REJECTED** | verbatim-Marlin: **correct but -39%**; offline-repack: workflow built (shared the GPU lock, `combined_definitive.sh:29`), same bf16-peak ceiling, no win | recorded verdict / combined_definitive.sh |
|
||||
|
||||
**Why the whole track loses (the structural reason):** bf16 tensor-core peak on
|
||||
GB10 is **~half FP4 peak** (PGR s3), so any dequant->bf16 kernel caps at ~half the
|
||||
throughput the native FP4-MMQ read reaches; and the dequant write is an
|
||||
un-amortized weight-sized memory pass (~8x the FP4-read byte traffic, PGR). The
|
||||
W4A16 angle was the most promising because it *also* erases the ~8% act-quant tax
|
||||
vLLM never pays - but the bf16-peak ceiling still made it a net regression. **MMQ
|
||||
is optimal; the GEMM bucket is not winnable on GB10 with the available kernels.**
|
||||
|
||||
### 2b. PREFILL - GDN chunked-scan track (verdict: M5 tf32 C=16 is the shipped winner)
|
||||
|
||||
The gated-DeltaNet chunked scan is the **#1 single prefill-gap contributor**
|
||||
(+59.2 us/tok, ~30% of the gap; LMAP). vLLM's FLA `chunk_gated_delta_rule` runs the
|
||||
same math at **36.5 us/tok vs paged 95.7 = 2.62x** (LMAP), pushing intra-chunk Gram
|
||||
products through tensor cores. The series chased that headroom.
|
||||
|
||||
| Lever | What | Verdict | Key number | Source |
|
||||
|---|---|---|---|---|
|
||||
| **0031** scalar-serial chunked scan | FLA-style chunk gated-delta-rule, scalar/serial form (`GDN_TC=0`) | superseded | math-correct (`test-backend-ops` 91/91, <=1e-7 NMSE) but **~761 vs ~971 t/s = ~22% slower** at the GB10-forced C=16 | README s5 |
|
||||
| **0047 / M5** tf32 tensor-core scan | full form-T solve + state-update on tf32 `m16n8k8` mma, f32-only re-port | **SHIPPED (default-on under paged)** | MoE prefill S_PP **+3.5% @npp512 (3x A/B), +17.7% @npp2048**; decode unchanged; bit-exact-benign (`GATED_DELTA_NET` 46-94/94, md5 == canonical) | README s3/s5 |
|
||||
| bf16 CONFIG-C (M8) | bf16 `Kc/Qc` + 2 C*C scratch, C->64 + 2 blk/SM | **REJECTED** (not in f32-only series) | the run that confirmed the geometry (CDEF GIT_HEAD), then dropped | CDEF / README s5 |
|
||||
| bf16-C16 | bf16 Gram at C=16 | rejected | no win over tf32-M5; bf16 mantissa unsafe on the state-coupled products | GDN build-plan s4 |
|
||||
| BV block-occupancy A/B (tf32) | raise blocks/SM to test if occupancy is the bound | **REJECTED** (occupancy is NOT the bound; latency is wave-hidden) | two arms statistically equal: **1844 vs 1814 S_PP (-1.04%, within noise)** | GDNAB armA/armB |
|
||||
| bf16-C64 | bf16 Gram at the larger C=64 chunk | **REJECTED** | **-18.75%** - the O(C^2) intra-chunk triangular-solve + serial recurrence dominates, so growing C hurts | recorded verdict / GDN build-plan |
|
||||
| Phase 10 C32 slab M5 | C=32 with two `dv_tile=64` slabs, default-off `GDN_C32_SLAB=1` | **REJECTED** | md5-clean after tail-row zeroing, but S_PP regressed: MoE 2048 **2430.32 -> 2054.86**, dense 2048 **1019.25 -> 903.73** | phase10 gates/ab |
|
||||
| Phase 11 QS-early M5 | move `QS = Qc * S0` earlier, default-off `GDN_M5_QS_EARLY=1` | **REJECTED** | md5-clean, but S_PP regressed slightly: MoE 2048 **2441.54 -> 2420.26**, dense 2048 **1021.06 -> 1015.77** | phase11 gates/ab |
|
||||
| Phase 12 shared-A/Ai cost model | f32 Ai scratch shared across two C32 value slabs | **GO to one prototype** | BT32 f32 scratch at npp2048,npl32: MoE 256 MiB / 768 MiB Ai traffic; dense 384 MiB / 1152 MiB Ai traffic | phase12 cost model |
|
||||
| Phase 13 Global-Ai32 | precompute f32 Ai once, consume from two C32 `dv_tile=64` slabs | **REJECTED** | md5-clean, but S_PP regressed: MoE 2048 **2425.10 -> 2097.76**, dense 2048 **1016.14 -> 918.19** | phase13 gates/ab |
|
||||
|
||||
**Why the bottleneck is not occupancy/dtype:** the cost is the **O(C^2)
|
||||
intra-chunk triangular solve + the serial inter-chunk recurrence dependency**, not
|
||||
grid occupancy (BV: -1.04%, latency is wave-hidden) and not Gram dtype (bf16-C64:
|
||||
-18.75%). GB10's 99 KB
|
||||
dynamic-smem cap forces **C=16** (the 128x128 f32 state alone is 64 KB of the
|
||||
all-shared layout), and at this head dim the only win is tensor cores on the
|
||||
intra-chunk products, not chunking or wider chunks. M5 tf32 at C=16 is exactly
|
||||
that and is the shipped winner; it does not fully close the 2.62x because vLLM's
|
||||
mature FLA blocked-solve is a more complete tensor-core implementation.
|
||||
|
||||
Post-record caveat closed: Phase 13 tested the one permitted
|
||||
`GDN_GLOBAL_AI32=1` prototype. It was correctness-clean but slower, so GDN kernel
|
||||
work on GB10 should stop rather than moving to f16 Ai or additional local
|
||||
reorders.
|
||||
|
||||
### 2c. DECODE / serving (verdict: near-parity at ~86% of vLLM's true GPU-steady decode; the earlier "BW-floored / vLLM pays equally" was a profiling artifact)
|
||||
|
||||
**Methodology correction - why every earlier decode decomposition was wrong.**
|
||||
Decode runs as a **replayed CUDA graph**. `nsys` *without* `--cuda-graph-trace=node`
|
||||
collapses each graph replay into a **single opaque launch**, so the per-kernel
|
||||
attribution in every prior decode profile (the "paged 159 us/tok, GPU ~16% busy,
|
||||
host-bound, 5.4x more GPU-efficient per token" picture, and the conclusion that the
|
||||
high-N gap was a pure bandwidth floor vLLM pays equally) was an **artifact of graph
|
||||
collapse, not real per-token cost**. The correct method, used for the numbers below
|
||||
(**HNP**, clean uncontended node, 2026-06-30), is `nsys --cuda-graph-trace=node`
|
||||
plus the **difference method**: per-token cost = the ntg=64 profile minus the
|
||||
ntg=16 profile, isolating per-token-linear work from fixed per-step overhead. Under
|
||||
this method **paged decode at npl=256 is 99% GPU-busy (GPU-idle only 1.4%), NOT
|
||||
host-bound** - the opposite of the collapsed-graph reading. This supersedes the
|
||||
LMAP decode decomposition.
|
||||
|
||||
**The real per-token decomposition (paged, npl=256, HNP)** - GPU-steady ~1082
|
||||
us/tok (924 t/s):
|
||||
|
||||
| Bucket | us/tok | % of decode | Note |
|
||||
|---|---:|---:|---|
|
||||
| GDN recurrent scan | 553 | **51%** | **LINEAR in batch** - the dominant cost; shared BW floor (below) |
|
||||
| NVFP4 expert GEMM | 254 | 23% | amortizes with batch |
|
||||
| bf16 projections | 73 | 7% | |
|
||||
| elementwise | 57 | 5% | |
|
||||
| SSM conv | 31 | 3% | |
|
||||
| rest | small | - | |
|
||||
| GPU-idle | - | **1.4%** | not host-bound |
|
||||
|
||||
**The gap reconciled (the numbers must sum).** The headline N=256 figures (perseq
|
||||
~56%, decode_agg ~61%, section 1b) were paged-**server** **718** over vLLM-**server**
|
||||
**1177**. But the vLLM server number is **inflated ~8 pts**: vLLM's true GPU-steady
|
||||
decode is **1078 t/s**, and its chunked-prefill overlap inflates the
|
||||
server-measured decode window. The reconciled chain:
|
||||
|
||||
| Measurement | t/s | % of vLLM-server (1177) |
|
||||
|---|---:|---:|
|
||||
| vLLM server (CDEF) | 1177 | 100% |
|
||||
| vLLM **true GPU-steady** decode | 1078 | 92% |
|
||||
| llama **GPU-steady** decode | 924 | 78.5% (**= 86% of vLLM's true 1078**) |
|
||||
| llama server (CDEF) | 718 | ~60.7% (61%) |
|
||||
|
||||
So **vs vLLM's true GPU-steady decode, paged is ~86%, not ~56%.** The ~56% headline
|
||||
conflated two distinct things: vLLM's prefill-overlap-inflated server window, and
|
||||
the paged serving graph-reuse overhead. The **~17 pt** drop from llama GPU-steady
|
||||
(78.5%) to llama server (60.7%) is exactly that **serving graph-reuse overhead**,
|
||||
which is **S3-recoverable** (2d).
|
||||
|
||||
**GDN is a shared BW floor where paged is ahead.** The GDN recurrent scan moves
|
||||
**~32 GB/step of f32 recurrent-state traffic**; paged runs it at **83% of the
|
||||
273 GB/s LPDDR5x peak vs vLLM's 79%**. Both engines' high-N sublinearity (only
|
||||
**1.17-1.18x throughput for a 2x batch**) comes from this **shared** floor - it is
|
||||
not a paged-specific loss, and paged is the faster of the two on it.
|
||||
|
||||
**The residual ~14 pt GPU-steady gap is real but not cheaply closable.** vLLM's
|
||||
GPU-steady 1078 vs paged 924 decomposes into two buckets: the **MoE expert path
|
||||
(~+11 ms)** - vLLM's fused Marlin persistent-tiling vs ggml's separate act-quant +
|
||||
MMQ - and **elementwise (~+10 ms)** - vLLM fuses it into one Triton kernel. Both
|
||||
fusions were attempted and rejected (table below). Closing the residual needs
|
||||
vLLM's mature Marlin tiling (our own ggml Marlin port already lost **-19.6%**) plus
|
||||
multi-stream overlap (hard inside a single-stream CUDA graph): **low-EV,
|
||||
multi-week, GB10-uncertain**.
|
||||
|
||||
**Decode / fusion levers (verdicts).**
|
||||
|
||||
| Lever | What | Verdict | Key number | Source |
|
||||
|---|---|---|---|---|
|
||||
| act-quant folded into ggml MMQ | erase the act-quant pass by quantizing the y-operand inside the MoE expert MMQ kernel (vLLM's fused-Marlin single-pass shape) | **REJECTED** (regression) | **-79.4%**: ggml MMQ re-quantizes the y-operand **once per weight-row-tile x stream-k split**, with no tensor cores for the inline quant - structural, ggml MMQ lacks vLLM's persistent single-pass tiling | HNP / recorded verdict |
|
||||
| norm + quant + silu fusion | fold the elementwise path into one launch (vLLM's Triton kernel) | **REJECTED** (architecturally infeasible) | `ggml_cuda_can_fuse` cannot express it: FP4 quant is a **mul_mat-internal prologue, not a cgraph node**; the norm is already fused (0042/0044); silu is separated from the norm by **2 GEMMs + the router** | recorded verdict |
|
||||
| Q8_0 / FP8 projection | quantize the bf16 GDN/attn projections (premise: vLLM uses FP8 here) | **REJECTED** (regime error, not premise error) | vLLM **does** use FP8 projections (confirmed from `hf_quant_config.json` `MIXED_PRECISION`), but at N=128/256 projections are only **~12% of the decode stream**, so this closes **<=6%, not the gap** | HNP / hf_quant_config.json |
|
||||
| NVFP4 the bf16 GDN/attn projections | drop projections to NVFP4 (more aggressive than FP8) | **REJECTED** | **KL-fail, ~+6% PPL**; vLLM keeps the SAME bf16/FP8 projections, never NVFP4 | LMAP |
|
||||
| W4A16-Marlin MoE decode | Marlin grouped expert GEMM on the decode path | **REJECTED** | BW-floored wash, **~5% slower** kernel | LMAP |
|
||||
| bf16-tau per-head SSM (0026) | per-head bf16 tau on the SSM decode | **DROPPED** | flat **780.6 vs 780.0 t/s** once the fusion patches landed | README s5 |
|
||||
| D3 FA-split / D4 GDN-width-adaptive | the older "off critical path" decode levers | **SUPERSEDED reasoning** | originally rejected via the now-debunked "5.4x faster / host-bound" reading; under HNP the GDN scan **is** the critical path (51%), but it is the shared BW floor where paged already leads (83% vs 79%), so neither is a win | HNP |
|
||||
|
||||
**Dense decode is AHEAD at low N (116.7% @ N=8, CDEF)** because the GPU is
|
||||
underutilized there and the paged path's per-token efficiency wins; this is the one
|
||||
operating point where paged is unambiguously faster than vLLM.
|
||||
|
||||
### 2d. SERVING / engine (verdict: host loop and scheduler closed; spec-decode orthogonal)
|
||||
|
||||
| Lever | What | Verdict | Key number | Source |
|
||||
|---|---|---|---|---|
|
||||
| **0040 / S1** paged decode-graph reuse | correct `can_reuse` keyed on bucketed block-table dims | **SHIPPED (default-on)** | serving graph reuse **0% -> 72.2%** (with S3); static **0% -> 95.5%** | README, DSS |
|
||||
| **0041 / S3** decode-shape-stable scheduling (`LLAMA_PAGED_DECODE_STABLE`) | keep prefill out of decode steps for reuse-stable shapes | **SHIPPED default-OFF** (opt-in throughput-max knob) | recovers the **~17 pt serving graph-reuse overhead** (llama server 60.7% -> toward GPU-steady 78.5%, 2c) at a TTFT cost; default-on regressed real serving: **2.5x worse TTFT** (60s vs 24s @N=256), **20-29% lower** end-to-end throughput, hence opt-in | README, DSS, HNP |
|
||||
| **0043 / D1** full-step MoE decode CUDA graph | graph the whole decode step incl. grouped-MMQ MoE dispatch | **SHIPPED (default-on)** | +2.6% (npl128) to +5-13% (npl32); the D1 premise "host-sync on MoE-routing readback" was **REFUTED** (sync count identical graphs on/off; 99% GPU-busy static) | README s5 |
|
||||
| S2 double-buffer set_inputs | overlap host input build with GPU | **DROPPED** | `set_inputs` is **~0.05 ms/step** - nothing to recover (the rebuild was the cost) | DSS |
|
||||
| whole-step graph / host loop | the host scheduling loop as the serving residual | **CLOSED (~0-1%)** | baseline reuse 0% (agg 757.6) **statistically equal** to S1+S3 reuse 72% (agg 763.3); `hostproc` only ~4-8% of the per-step wall = **measured dead** | DSS |
|
||||
| padded / fixed-slot decode | pad decode width to `--parallel` for ~100% reuse | **REJECTED (built, GPU-tested)** | inert (md5 bit-exact) but **regresses at every concurrency**; N=8 burst 28.16 -> 6.05 tok/s/seq (~4.6x slower); serving decode is **GPU-compute-bound**, dummy-row compute > reuse recovered | DSS |
|
||||
| speculative decode (MTP) | draft + verify; greedy is bit-exact | **REJECTED for current GB10 serving** | Phase 14 passed safety, but Phase 15 direct serving A/B regressed at every tested concurrency (n128 decode agg 662.4 -> 138.5 tok/s) despite high acceptance; Phase 16 profile supports graph-reuse loss as root cause (`graphs reused` 62 -> 1 in the small nsys run). Not a parity lever unless a future graph/batch-shape fix changes this result | LMAP |
|
||||
|
||||
The serving regime was the one place the static-bench parity did not carry over
|
||||
(paged ~3.7 vs vLLM ~5.9 tok/s/seq, -39%, DSS). S1 made the decode step reusable
|
||||
and the host loop was driven to ~0-1% of the wall. The graph-node-traced HNP
|
||||
profile (2c) then resolves the remaining serving gap into two parts: the **~17 pt
|
||||
serving graph-reuse overhead** (S3-recoverable via this knob) and the **~14 pt
|
||||
GPU-steady kernel gap** vs vLLM's true 1078 t/s (vLLM's fused-Marlin MoE + Triton
|
||||
elementwise, 2c). Both are real; neither is the "pure LPDDR5x floor, vLLM pays
|
||||
equally" story the collapsed-graph profile implied.
|
||||
|
||||
---
|
||||
|
||||
## 3. Structural floors (not closable on GB10)
|
||||
|
||||
These are the hardware/algorithm ceilings the investigation hit. They are why
|
||||
parity is unreachable on this part, and they are the levers' "why" in one place.
|
||||
|
||||
1. **LPDDR5x bandwidth (~273 GB/s) bounds the GDN recurrent scan - a *shared*
|
||||
floor where paged leads.** The GDN scan is the dominant decode bucket (553
|
||||
us/tok, 51%, LINEAR in batch; HNP) and moves ~32 GB/step of f32 recurrent
|
||||
state; paged runs it at **83% of the 273 GB/s peak vs vLLM's 79%**, and both
|
||||
engines' high-N sublinearity (1.17-1.18x for a 2x batch) is this same floor.
|
||||
This is **not** the explanation for the high-N server-window gap: the
|
||||
graph-node-traced HNP profile (2c) shows paged decode **99% GPU-busy at ~86% of
|
||||
vLLM's true GPU-steady decode**, with the server-window ~56% being a
|
||||
prefill-overlap measurement artifact (~8 pt) plus an S3-recoverable graph-reuse
|
||||
overhead (~17 pt), not a bandwidth floor vLLM pays equally. The residual ~14 pt
|
||||
GPU-steady gap is kernel maturity (point 4 below + 2c), not bandwidth. On
|
||||
datacenter HBM (B200: ~8 TB/s) this GDN floor lifts ~30x.
|
||||
|
||||
2. **FP4-MMQ optimality at GB10's tensor-core ratios.** Native FP4-MMQ at M<=128 is
|
||||
at the FP4 weight-BW floor (decode) and beats every dequant->bf16 alternative at
|
||||
large M (prefill), because bf16 TC peak is ~half FP4 peak on sm_121 and the
|
||||
dequant pass is an un-amortized memory pass (PGR). vLLM itself is on a **bf16
|
||||
Marlin fallback** here (no tcgen05/CUTLASS-grouped FP4 on consumer Blackwell,
|
||||
CUTLASS #3096), so there is no faster GEMM to port.
|
||||
|
||||
3. **GDN O(C^2) intra-chunk solve + serial inter-chunk recurrence.** The chunked
|
||||
scan's cost is the triangular A-inverse solve (quadratic in chunk size C) plus
|
||||
the strictly-serial cross-chunk state carry, with C forced to 16 by the 99 KB
|
||||
smem cap. Occupancy (BV: -1%) and dtype (bf16-C64: -18.75%) are not the bound;
|
||||
only a fuller tensor-core blocked-solve closes the residual 2.62x, and M5 tf32
|
||||
captures the tractable part.
|
||||
|
||||
4. **vLLM's mature fused kernels (FLA blocked-solve, fused-Marlin MoE, Triton
|
||||
elementwise) are tuned for HBM.** They are the source of both the prefill cap
|
||||
and the residual ~14 pt decode GPU-steady gap (2c): the fused-Marlin
|
||||
persistent-tiling MoE path (~+11 ms) and the single-kernel Triton elementwise
|
||||
(~+10 ms). The matching ggml fusions were rejected as infeasible or regressive
|
||||
(2c): folding act-quant into MMQ regressed -79.4% (no single-pass tiling), and
|
||||
norm+quant+silu cannot be expressed via `ggml_cuda_can_fuse`. The FLA chunked
|
||||
GDN, Marlin grouped GEMM, and FULL/PIECEWISE cudagraphs all assume datacenter
|
||||
bandwidth and TC ratios; they are real wins on B200, which is why closing the
|
||||
residual is a different-hardware question (mature kernels + multi-stream
|
||||
overlap), not a missing single-lever optimization.
|
||||
|
||||
---
|
||||
|
||||
## 4. Shipped wins (all bit-exact / KL-benign)
|
||||
|
||||
What the series actually banks, all gated per-path:
|
||||
|
||||
- **FP4-MMQ MoE/dense GEMM** - native Blackwell FP4-MMA, at the FP4 weight-BW
|
||||
floor (decode parity) and beating every dequant alternative at prefill. The
|
||||
reason the whole 2a track stays default-off.
|
||||
- **M5 tf32 tensor-core chunked GDN prefill (patch 0047)** - default-on under
|
||||
`LLAMA_KV_PAGED`; MoE prefill **+3.5% @npp512, +17.7% @npp2048**, decode
|
||||
untouched, bit-exact-benign.
|
||||
- **0042 fused residual-add + RMSNorm + weight-mul** - one kernel for `h = x +
|
||||
sub; n = rms_norm(h) * w`; dense S_PP +0.5%, bit-exact.
|
||||
- **0044 fused gated RMSNorm + SiLU gate-mul (GatedRMSNorm fusion)** - the GDN
|
||||
output norm `(rms_norm(x)*w)*silu(z)` folded into one launch (672 -> 336
|
||||
launches @npp512); S_PP dense +1.1%, MoE +0.9%, `test-backend-ops` 12979/12979.
|
||||
- **0046 GDN-prefill geometry gate** - gates patch 0022's decode occupancy retune
|
||||
by scan length so it stops regressing dense prefill; recovers **+7.2%** dense
|
||||
prefill back to stock parity while keeping the decode win, bit-exact.
|
||||
- **SSM decode fusion stack (0018-0022, 0028)** - in-place state, fused gather,
|
||||
o_proj MMQ reshape, conv in-place, occupancy retune; the **2.26x/2.46x over
|
||||
stock** decode multiplier (README).
|
||||
- **Serving host loop closed (0040 S1, 0043 D1)** - decode-graph reuse and
|
||||
full-step graph capture; host loop driven to ~0-1% of the serving wall.
|
||||
- **The memory advantage** - **1.5-3x lower VRAM** than vLLM (NVFP4-resident, no
|
||||
persistent bf16 dequant copies; CDEF PEAK_GB e.g. MoE N=8 50 vs 112 GB), which
|
||||
is a legitimate higher-max-concurrency-per-GPU operating point.
|
||||
- **Low-N decode efficiency** - dense decode **ahead of vLLM (116.7% @ N=8)**.
|
||||
- **Bit-exact output** - per-path greedy md5 stable (dense `5951a5b4`, paged-MoE
|
||||
`8cb0ce23`), the sacred gate held through the entire series.
|
||||
|
||||
---
|
||||
|
||||
## 5. The parity verdict and the path
|
||||
|
||||
**Verdict (revised): PREFILL is genuinely capped on GB10; DECODE-SERVING is near
|
||||
vLLM parity (~86% of its true GPU-steady decode), with the long-standing ~56%
|
||||
headline now identified as a measurement / operating-point artifact.** Prefill
|
||||
sits at **36% (MoE) / 43% (dense)** of vLLM and is a real floor (FP4-MMQ optimality
|
||||
+ GDN O(C^2) intra-chunk complexity; prefill is **not** CUDA-graph-replayed, so
|
||||
unlike decode these numbers are not profiling artifacts). The GDN chunked scan is
|
||||
at its tractable tensor-core win (M5) and the prefill GEMM bucket is FP4-MMQ-optimal
|
||||
(every alternative rejected; vLLM is itself on a bf16-Marlin fallback here). For
|
||||
decode, the graph-node-traced HNP profile corrects the record: paged decode is
|
||||
**99% GPU-busy at ~86% of vLLM's true GPU-steady decode (924 vs 1078 t/s)**; the
|
||||
~56% server-window figure was vLLM's prefill-overlap inflation (~8 pt) plus the
|
||||
S3-recoverable serving graph-reuse overhead (~17 pt). The residual **~14 pt**
|
||||
GPU-steady gap is vLLM's mature fused-Marlin MoE (~+11 ms) and Triton elementwise
|
||||
(~+10 ms) kernels; the matching ggml fusions were rejected (act-quant-into-MMQ
|
||||
-79.4%, norm+quant+silu infeasible), and closing the residual needs mature Marlin
|
||||
tiling (our port lost -19.6%) plus multi-stream overlap - low-EV, multi-week,
|
||||
GB10-uncertain, not a free bit-exact lever.
|
||||
|
||||
**The honest framing:** on GB10 the paged backend is **at or ahead of vLLM at low
|
||||
concurrency (dense 117% @N=8), uses 1.5-3x less memory, and is bit-exact**, runs
|
||||
high-N decode at **~86% of vLLM's true GPU-steady decode** (the ~56% server-window
|
||||
number is a measurement artifact, 2c), and sits at **~36% (MoE) / ~43% (dense) of
|
||||
vLLM prefill**. The prefill residual is a real FP4-MMQ + GDN-O(C^2) floor; the
|
||||
~14 pt decode residual is vLLM's mature fused kernels, not engineering debt and not
|
||||
a cheap lever.
|
||||
|
||||
**The path to parity is different hardware.** A datacenter Blackwell (B200,
|
||||
~8 TB/s HBM, native tcgen05/CUTLASS FP4, TMEM) lifts the bandwidth floor ~30x and
|
||||
**restores exactly the vLLM advantages that lose on GB10**: its FLA blocked-solve
|
||||
GDN, its Marlin/CUTLASS grouped FP4 GEMM, and its HBM-tuned full-cudagraph decode
|
||||
all assume that bandwidth and those TC ratios. On that hardware the parity question
|
||||
is re-opened from scratch; on GB10 it is closed. Do not re-litigate the GB10 levers
|
||||
- re-run the methodology on the new silicon instead.
|
||||
|
||||
---
|
||||
|
||||
*Recorded per `.agents/vllm-parity-methodology.md` (both-engine ground-truth,
|
||||
per-lever A/B, record-rejected-levers). All GPU numbers from `ssh dgx.casa`
|
||||
artifacts under `~/bench/`; all in-repo numbers from the docs cited in the source
|
||||
key. The GPU lock was not touched in producing this document (CPU-only:
|
||||
artifact-read + write).*
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,25 +0,0 @@
|
||||
model,engine,npl,decode_agg_tps,prefill_tps
|
||||
q36-27b-nvfp4,llama-stock,8,68.3,937.7
|
||||
q36-27b-nvfp4,llama-stock,32,119.9,885.2
|
||||
q36-27b-nvfp4,llama-stock,64,142.8,885.1
|
||||
q36-27b-nvfp4,llama-stock,128,155.1,887.2
|
||||
q36-27b-nvfp4,llama-patched,8,85.3,915.1
|
||||
q36-27b-nvfp4,llama-patched,32,211.9,919.0
|
||||
q36-27b-nvfp4,llama-patched,64,305.2,923.5
|
||||
q36-27b-nvfp4,llama-patched,128,382.1,922.9
|
||||
q36-27b-nvfp4,vllm,8,70.4,2096.2
|
||||
q36-27b-nvfp4,vllm,32,211.8,2182.6
|
||||
q36-27b-nvfp4,vllm,64,309.1,2088.9
|
||||
q36-27b-nvfp4,vllm,128,418.8,1929.1
|
||||
q36-35b-a3b-nvfp4,llama-stock,8,186.7,1501.5
|
||||
q36-35b-a3b-nvfp4,llama-stock,32,267.4,1856.8
|
||||
q36-35b-a3b-nvfp4,llama-stock,64,320.5,1949.5
|
||||
q36-35b-a3b-nvfp4,llama-stock,128,347.2,1995.4
|
||||
q36-35b-a3b-nvfp4,llama-patched,8,230.3,1510.3
|
||||
q36-35b-a3b-nvfp4,llama-patched,32,466.4,1969.2
|
||||
q36-35b-a3b-nvfp4,llama-patched,64,622.4,2122.8
|
||||
q36-35b-a3b-nvfp4,llama-patched,128,784.3,2177.0
|
||||
q36-35b-a3b-nvfp4,vllm,8,256.5,5186.5
|
||||
q36-35b-a3b-nvfp4,vllm,32,500.8,6223.4
|
||||
q36-35b-a3b-nvfp4,vllm,64,686.1,5926.5
|
||||
q36-35b-a3b-nvfp4,vllm,128,882.2,5300.5
|
||||
|
@@ -1,217 +0,0 @@
|
||||
// Paged-pool burst-degradation repro (patch 0024). DEV SCAFFOLDING ONLY.
|
||||
//
|
||||
// Reproduces, at the libllama level, the two host-side defects behind the
|
||||
// "later lower-npl prefill collapses, decode fine, restart cures it" benchmark
|
||||
// signature:
|
||||
//
|
||||
// * RECLAMATION GAP (Fix-1): a partial tail seq_rm(seq, p0>0, -1) - exactly
|
||||
// what llama-server issues on every reused slot - frees the kv-cache CELLS
|
||||
// but the paged manager keeps owning the trailing BLOCKS. The manager's
|
||||
// free pool silently shrinks. Test A measures the reclaimed-block delta.
|
||||
//
|
||||
// * FRAGMENTATION / NO COMPACTION (Fix-2): a high-fan-out burst that allocates
|
||||
// many sequences and frees them in a scrambled order leaves the free queue a
|
||||
// scrambled permutation of physical block ids. A later low-npl prefill then
|
||||
// pops physically scattered blocks, so its KV scatter-write + in-kernel
|
||||
// paged-attention gather lose locality and prefill throughput collapses;
|
||||
// decode (single-token append) barely notices. Test B times an npl8 prefill
|
||||
// on a FRESH pool vs an npl8 prefill AFTER a scrambling burst+drain.
|
||||
//
|
||||
// PASS (post-fix): Test A reclaims ceil((PP-KEEP)/bs) trailing blocks on the
|
||||
// partial seq_rm (0 pre-fix); Test B's post-burst npl8 prefill_tps is within ~10%
|
||||
// of the fresh npl8 and num_free returns to the pristine value after the drain.
|
||||
//
|
||||
// Run with LLAMA_KV_PAGED=1. Env: BURST_NSLOT(64) NPL(8) PP(512) KEEP(256)
|
||||
// GEN(4) PAGED_NGL(99). All sequences use distinct content so nothing is shared.
|
||||
|
||||
#include "llama.h"
|
||||
#include "paged-prefix-api.h"
|
||||
|
||||
#include <chrono>
|
||||
#include <clocale>
|
||||
#include <cstdio>
|
||||
#include <cstdlib>
|
||||
#include <cstring>
|
||||
#include <vector>
|
||||
|
||||
static int env_i(const char * k, int dflt) { const char * v = getenv(k); return v ? atoi(v) : dflt; }
|
||||
|
||||
using clk = std::chrono::steady_clock;
|
||||
static double secs(clk::time_point a, clk::time_point b) {
|
||||
return std::chrono::duration<double>(b - a).count();
|
||||
}
|
||||
|
||||
struct Ctx { llama_context * ctx; llama_memory_t mem; llama_batch batch; int n_vocab; };
|
||||
|
||||
// Deterministic, content-distinct token for (seq, pos): keeps every sequence's
|
||||
// blocks unique so no cross-request prefix sharing masks the accounting.
|
||||
static llama_token tok_of(int seq, int pos, int n_vocab) {
|
||||
return (llama_token) (((seq * 1000003 + pos * 131 + 7) % (n_vocab - 200)) + 100);
|
||||
}
|
||||
|
||||
// Prefill n tokens of seq at [pos0, pos0+n) in one ubatch (n <= n_batch).
|
||||
// Returns wall seconds (sync'd).
|
||||
static double prefill(Ctx & C, int seq, int pos0, int n) {
|
||||
clk::time_point t0 = clk::now();
|
||||
C.batch.n_tokens = 0;
|
||||
for (int j = 0; j < n; ++j) {
|
||||
int i = C.batch.n_tokens;
|
||||
C.batch.token[i] = tok_of(seq, pos0 + j, C.n_vocab);
|
||||
C.batch.pos[i] = pos0 + j;
|
||||
C.batch.n_seq_id[i] = 1;
|
||||
C.batch.seq_id[i][0]= seq;
|
||||
C.batch.logits[i] = (j + 1 == n) ? 1 : 0;
|
||||
C.batch.n_tokens++;
|
||||
}
|
||||
if (llama_decode(C.ctx, C.batch)) { fprintf(stderr, "prefill decode failed seq=%d\n", seq); return -1; }
|
||||
llama_synchronize(C.ctx);
|
||||
return secs(t0, clk::now());
|
||||
}
|
||||
|
||||
// One decode step (single token) for seq at pos.
|
||||
static void decode1(Ctx & C, int seq, int pos) {
|
||||
C.batch.n_tokens = 1;
|
||||
C.batch.token[0] = tok_of(seq, pos, C.n_vocab);
|
||||
C.batch.pos[0] = pos; C.batch.n_seq_id[0] = 1; C.batch.seq_id[0][0] = seq; C.batch.logits[0] = 1;
|
||||
if (llama_decode(C.ctx, C.batch)) fprintf(stderr, "decode1 failed seq=%d\n", seq);
|
||||
}
|
||||
|
||||
int main(int argc, char ** argv) {
|
||||
std::setlocale(LC_NUMERIC, "C");
|
||||
const char * model_path = nullptr;
|
||||
for (int i = 1; i < argc; ++i) if (!strcmp(argv[i], "-m") && i + 1 < argc) model_path = argv[++i];
|
||||
if (!model_path) { fprintf(stderr, "usage: %s -m model.gguf\n", argv[0]); return 2; }
|
||||
|
||||
const int NSLOT = env_i("BURST_NSLOT", 64);
|
||||
const int NPL = env_i("NPL", 8);
|
||||
const int PP = env_i("PP", 512);
|
||||
const int KEEP = env_i("KEEP", 256);
|
||||
const int GEN = env_i("GEN", 4);
|
||||
const int ngl = env_i("PAGED_NGL", 99);
|
||||
const bool paged = getenv("LLAMA_KV_PAGED") != nullptr;
|
||||
|
||||
ggml_backend_load_all();
|
||||
llama_model_params mp = llama_model_default_params();
|
||||
mp.n_gpu_layers = ngl;
|
||||
llama_model * model = llama_model_load_from_file(model_path, mp);
|
||||
if (!model) { fprintf(stderr, "model load failed\n"); return 1; }
|
||||
const llama_vocab * vocab = llama_model_get_vocab(model);
|
||||
const int n_vocab = llama_vocab_n_tokens(vocab);
|
||||
|
||||
// Pool sized for the burst plus headroom so the burst fits but a later npl
|
||||
// run draws from whatever the burst's churn left behind.
|
||||
const long cells = (long) (NSLOT + NPL + 4) * (PP + GEN + 16);
|
||||
llama_context_params cp = llama_context_default_params();
|
||||
cp.n_ctx = (uint32_t) cells;
|
||||
cp.n_batch = (uint32_t) (PP + 16);
|
||||
cp.n_ubatch = (uint32_t) (PP + 16);
|
||||
cp.n_seq_max = NSLOT + NPL + 2;
|
||||
cp.kv_unified = true; // one unified stream-0 pool -> num_free(ctx) is the whole pool
|
||||
cp.no_perf = true;
|
||||
llama_context * ctx = llama_init_from_model(model, cp);
|
||||
if (!ctx) { fprintf(stderr, "ctx init failed (cells=%ld)\n", cells); return 1; }
|
||||
|
||||
Ctx C; C.ctx = ctx; C.mem = llama_get_memory(ctx); C.n_vocab = n_vocab;
|
||||
C.batch = llama_batch_init(cp.n_batch, 0, 1);
|
||||
|
||||
printf("== paged-burst-bench == paged=%d NSLOT=%d NPL=%d PP=%d KEEP=%d GEN=%d n_ctx=%ld\n",
|
||||
paged, NSLOT, NPL, PP, KEEP, GEN, cells);
|
||||
|
||||
llama_memory_clear(C.mem, true);
|
||||
const long F_start = paged_prefix_api::num_free_global();
|
||||
|
||||
// ---- Test A: Fix-1 reclamation gap on a partial tail seq_rm --------------
|
||||
{
|
||||
prefill(C, 0, 0, PP);
|
||||
const long f_after_prefill = paged_prefix_api::num_free_global();
|
||||
llama_memory_seq_rm(C.mem, 0, KEEP, -1); // partial tail removal
|
||||
const long f_after_rm = paged_prefix_api::num_free_global();
|
||||
llama_memory_seq_rm(C.mem, 0, -1, -1); // full free -> pristine
|
||||
const long f_after_full = paged_prefix_api::num_free_global();
|
||||
const long bs = 16;
|
||||
const long expect = (PP + bs - 1)/bs - (KEEP + bs - 1)/bs; // trailing blocks
|
||||
printf("[TEST-A Fix-1] start=%ld afterPrefill=%ld afterPartialRm=%ld reclaimed=%ld "
|
||||
"(expect %ld post-fix, 0 pre-fix) afterFullFree=%ld\n",
|
||||
F_start, f_after_prefill, f_after_rm, f_after_rm - f_after_prefill, expect, f_after_full);
|
||||
}
|
||||
|
||||
// ---- Test B: fragmentation -> npl prefill collapse -----------------------
|
||||
// Fresh npl prefill baseline on a pristine pool.
|
||||
llama_memory_clear(C.mem, true);
|
||||
double tps_fresh;
|
||||
{
|
||||
clk::time_point t0 = clk::now();
|
||||
long ntok = 0;
|
||||
for (int s = 0; s < NPL; ++s) { double d = prefill(C, s, 0, PP); if (d < 0) return 1; ntok += PP; }
|
||||
tps_fresh = ntok / secs(t0, clk::now());
|
||||
for (int s = 0; s < NPL; ++s) llama_memory_seq_rm(C.mem, s, -1, -1);
|
||||
}
|
||||
const long F_pristine = paged_prefix_api::num_free_global();
|
||||
|
||||
// High-fan-out burst: allocate NSLOT sequences, each prefilled + a few decode
|
||||
// steps (mixed alloc), then drain them in a scrambled order (odd ids first,
|
||||
// then even, each truncated before the full free) so the free queue becomes a
|
||||
// scrambled permutation - the fragmentation the bug never compacts.
|
||||
for (int s = 0; s < NSLOT; ++s) {
|
||||
if (prefill(C, NPL + s, 0, PP) < 0) return 1;
|
||||
for (int g = 0; g < GEN; ++g) decode1(C, NPL + s, PP + g);
|
||||
}
|
||||
const long F_during_burst = paged_prefix_api::num_free_global();
|
||||
// Drain: partial tail seq_rm (the reused-slot pattern) then full free, in a
|
||||
// scrambled slot order to scramble the physical free order.
|
||||
for (int parity = 1; parity >= 0; --parity)
|
||||
for (int s = 0; s < NSLOT; ++s) if ((s & 1) == parity) {
|
||||
llama_memory_seq_rm(C.mem, NPL + s, KEEP, -1); // partial (Fix-1 path)
|
||||
llama_memory_seq_rm(C.mem, NPL + s, -1, -1); // full free
|
||||
}
|
||||
const long F_after_drain = paged_prefix_api::num_free_global();
|
||||
|
||||
// Post-burst npl prefill: pops from the (pre-fix scrambled / post-fix
|
||||
// defragged) free queue.
|
||||
double tps_post;
|
||||
{
|
||||
clk::time_point t0 = clk::now();
|
||||
long ntok = 0;
|
||||
for (int s = 0; s < NPL; ++s) { double d = prefill(C, s, 0, PP); if (d < 0) return 1; ntok += PP; }
|
||||
tps_post = ntok / secs(t0, clk::now());
|
||||
for (int s = 0; s < NPL; ++s) llama_memory_seq_rm(C.mem, s, -1, -1);
|
||||
}
|
||||
|
||||
const double ratio = tps_fresh > 0 ? tps_post / tps_fresh : 0;
|
||||
printf("[TEST-B frag] num_free: start=%ld pristine=%ld duringBurst=%ld afterDrain=%ld "
|
||||
"(afterDrain==pristine? %s)\n",
|
||||
F_start, F_pristine, F_during_burst, F_after_drain,
|
||||
F_after_drain == F_pristine ? "YES" : "NO");
|
||||
printf("[TEST-B frag] prefill_tps fresh=%.1f post-burst=%.1f ratio=%.3f "
|
||||
"(PASS if >=0.90)\n", tps_fresh, tps_post, ratio);
|
||||
|
||||
// ---- Test C: idle-slot retention leak -> reclaim (the Fix-3 scenario) -----
|
||||
// Burst NSLOT sequences and leave them IDLE (stock llama-server keeps an idle
|
||||
// slot's KV; the blocks are stranded). F_idle shows the depleted pool a later
|
||||
// low-npl run would see. Then full-seq_rm each (exactly what Fix-3's
|
||||
// prompt_clear() issues at slot.release): F_reclaimed must return to pristine.
|
||||
llama_memory_clear(C.mem, true);
|
||||
// Touch the pool once so the manager exists, then read the full-pool size
|
||||
// (num_free is 0 while no manager is registered).
|
||||
if (prefill(C, 0, 0, 16) < 0) return 1;
|
||||
llama_memory_seq_rm(C.mem, 0, -1, -1);
|
||||
const long F_pre_c = paged_prefix_api::num_free_global();
|
||||
for (int s = 0; s < NSLOT; ++s) { if (prefill(C, NPL + s, 0, PP) < 0) return 1; }
|
||||
const long F_idle = paged_prefix_api::num_free_global();
|
||||
for (int s = 0; s < NSLOT; ++s) llama_memory_seq_rm(C.mem, NPL + s, -1, -1); // Fix-3 release
|
||||
const long F_reclaimed = paged_prefix_api::num_free_global();
|
||||
printf("[TEST-C idle] pristine=%ld idle_after_burst=%ld (leaked=%ld) reclaimed=%ld "
|
||||
"(returns_to_fresh? %s)\n",
|
||||
F_pre_c, F_idle, F_pre_c - F_idle, F_reclaimed,
|
||||
F_reclaimed == F_pre_c ? "YES" : "NO");
|
||||
|
||||
printf("RESULT paged=%d frag_fix2_ratio=%.3f drain_numfree_returns=%s idle_reclaim_returns=%s\n",
|
||||
paged, ratio,
|
||||
F_after_drain == F_pristine ? "YES" : "NO",
|
||||
F_reclaimed == F_pre_c ? "YES" : "NO");
|
||||
|
||||
llama_batch_free(C.batch);
|
||||
llama_free(ctx);
|
||||
llama_model_free(model);
|
||||
return 0;
|
||||
}
|
||||
@@ -1,59 +0,0 @@
|
||||
// Host-side unit test for the paged-pool burst-reclaim fix (patch 0024).
|
||||
// Compiles paged-kv-manager.cpp directly; no ggml / llama / GPU dependency.
|
||||
//
|
||||
// Fix-1 PagedKVManager::truncate(seq, n_keep) reclaims the trailing blocks
|
||||
// beyond ceil(n_keep/bs) (ref-counted), so a partial tail seq_rm no
|
||||
// longer strands blocks whose cells were cleared.
|
||||
// Fix-2 defrag_free_pool() relinks the free queue into ascending block-id
|
||||
// order once the pool is fully idle, undoing a burst's scrambled frees
|
||||
// so a later prefill pops physically contiguous blocks again.
|
||||
|
||||
#include "paged-kv-manager.h"
|
||||
#include <cstdio>
|
||||
|
||||
using paged::PagedKVManager;
|
||||
|
||||
int main() {
|
||||
int rc = 0;
|
||||
|
||||
// ---- Fix-1: truncate reclaims the trailing block suffix -----------------
|
||||
{
|
||||
PagedKVManager m(/*num_blocks=*/64, /*block_size=*/16, /*caching=*/true);
|
||||
const size_t f0 = m.num_free_blocks(); // 63 (block 0 reserved as null)
|
||||
m.allocate(0, 512); // ceil(512/16)=32 blocks
|
||||
const size_t f1 = m.num_free_blocks(); // 31
|
||||
m.truncate(0, 256); // keep ceil(256/16)=16, free 16
|
||||
const size_t f2 = m.num_free_blocks(); // 47
|
||||
printf("[unit Fix-1] free=%zu alloc512=%zu truncate256=%zu reclaimed=%zu (expect 16)\n",
|
||||
f0, f1, f2, f2 - f1);
|
||||
if (f2 - f1 != 16) rc = 1;
|
||||
m.truncate(0, 16); // keep 1 block, free 15 more
|
||||
const size_t f3 = m.num_free_blocks(); // 62
|
||||
printf("[unit Fix-1] truncate16=%zu (expect %zu)\n", f3, f0 - 1);
|
||||
if (f3 != f0 - 1) rc = 1;
|
||||
m.free(0);
|
||||
if (m.num_free_blocks() != f0) { printf("[unit Fix-1] free mismatch\n"); rc = 1; }
|
||||
}
|
||||
|
||||
// ---- Fix-2: defrag restores ascending popleft order ---------------------
|
||||
{
|
||||
PagedKVManager m(/*num_blocks=*/64, /*block_size=*/16, /*caching=*/false);
|
||||
for (int s = 0; s < 8; ++s) m.allocate(s, 16); // pop blocks 1..8
|
||||
const int scrambled[8] = {3, 7, 1, 5, 0, 6, 2, 4}; // free out of order
|
||||
for (int i = 0; i < 8; ++i) m.free(scrambled[i]);
|
||||
m.defrag_free_pool(); // all idle -> compact
|
||||
m.allocate(100, 16 * 3); // pop 3 blocks
|
||||
const auto bt = m.block_table(100);
|
||||
bool asc = true;
|
||||
printf("[unit Fix-2] post-defrag block_table:");
|
||||
for (size_t i = 0; i < bt.size(); ++i) {
|
||||
printf(" %d", bt[i]);
|
||||
if (i && bt[i] < bt[i - 1]) asc = false;
|
||||
}
|
||||
printf(" ascending=%s (expect YES)\n", asc ? "YES" : "NO");
|
||||
if (!asc) rc = 1;
|
||||
}
|
||||
|
||||
printf("UNIT %s\n", rc == 0 ? "PASS" : "FAIL");
|
||||
return rc;
|
||||
}
|
||||
Binary file not shown.
|
Before Width: | Height: | Size: 217 KiB |
Binary file not shown.
|
Before Width: | Height: | Size: 123 KiB |
Binary file not shown.
|
Before Width: | Height: | Size: 125 KiB |
@@ -1,439 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
usage() {
|
||||
cat <<'EOF'
|
||||
Usage: paged-current-serving-snapshot.sh [--summarize-gates ART]
|
||||
|
||||
Run a current-stack paged llama.cpp vs vLLM MoE serving snapshot on DGX.
|
||||
|
||||
This harness uses the clean llama.cpp mirror by default, not stale development
|
||||
trees. It runs pre/post paged inference gates, then a same-session serving
|
||||
comparison with the h2h client.
|
||||
|
||||
Environment overrides:
|
||||
SRC llama.cpp source dir (default: ~/llama-phase6-source)
|
||||
BUILD_DIR llama.cpp CMake build dir (default: $SRC/build-cuda)
|
||||
BIN llama.cpp build bin dir (default: $SRC/build-cuda/bin)
|
||||
MODEL paged GGUF path (default: ~/bench/q36-35b-a3b-nvfp4.gguf)
|
||||
VLLM_MODEL vLLM model dir (default: ~/bench/q36-35b-a3b-nvfp4-vllm)
|
||||
SERVED_MODEL_NAME OpenAI model name used by llama-server, vLLM, and h2h (default: q36)
|
||||
H2H h2h client (default: ~/bench/h2h_cli3.py)
|
||||
ART artifact dir (default: ~/bench/phase_current_serving_snapshot/<timestamp>)
|
||||
NPL concurrency list (default: "8 32 128")
|
||||
PTOK prompt filler words (default: 128)
|
||||
GEN generated tokens (default: 64)
|
||||
CTX llama-server context (default: 131072)
|
||||
PARALLEL llama-server parallel slots (default: 128)
|
||||
BATCH llama-server logical batch (default: 2048)
|
||||
UBATCH llama-server physical batch (default: 512)
|
||||
LLAMA_PORT llama-server port (default: 8098)
|
||||
LLAMA_READY_ATTEMPTS llama-server readiness attempts, one per second (default: 240)
|
||||
VLLM_PORT vLLM port (default: 8000)
|
||||
VLLM_BIN vLLM executable (default: ~/vllm-bench/bin/vllm)
|
||||
VLLM_READY_ATTEMPTS vLLM readiness attempts, one per second (default: 600)
|
||||
VLLM_GPU_MEMORY_UTILIZATION vLLM --gpu-memory-utilization (default: 0.85)
|
||||
VLLM_MAX_MODEL_LEN vLLM --max-model-len (default: 4096)
|
||||
VLLM_MAX_NUM_SEQS vLLM --max-num-seqs (default: 256)
|
||||
VLLM_TENSOR_PARALLEL_SIZE vLLM --tensor-parallel-size (default: 1)
|
||||
VLLM_EXTRA_ARGS whitespace-split extra args appended to vLLM serve (default: empty)
|
||||
SKIP_GATES=1 to skip pre/post paged inference gates
|
||||
DRY_RUN=1 validate inputs/preflight, write hardware.txt, and print commands without running servers
|
||||
|
||||
Options:
|
||||
--summarize-gates ART write ART/gate_summary.tsv from existing gate_pre/gate_post artifacts
|
||||
EOF
|
||||
}
|
||||
|
||||
SUMMARY_GATES_ART=""
|
||||
case "${1:-}" in
|
||||
-h|--help)
|
||||
usage
|
||||
exit 0
|
||||
;;
|
||||
--summarize-gates)
|
||||
if [[ -z "${2:-}" ]]; then
|
||||
usage >&2
|
||||
exit 2
|
||||
fi
|
||||
SUMMARY_GATES_ART="$2"
|
||||
;;
|
||||
"")
|
||||
;;
|
||||
*)
|
||||
usage >&2
|
||||
exit 2
|
||||
;;
|
||||
esac
|
||||
|
||||
SRC=${SRC:-"$HOME/llama-phase6-source"}
|
||||
BUILD_DIR=${BUILD_DIR:-"$SRC/build-cuda"}
|
||||
BIN=${BIN:-"$BUILD_DIR/bin"}
|
||||
MODEL=${MODEL:-"$HOME/bench/q36-35b-a3b-nvfp4.gguf"}
|
||||
VLLM_MODEL=${VLLM_MODEL:-"$HOME/bench/q36-35b-a3b-nvfp4-vllm"}
|
||||
SERVED_MODEL_NAME=${SERVED_MODEL_NAME:-q36}
|
||||
H2H=${H2H:-"$HOME/bench/h2h_cli3.py"}
|
||||
ART=${ART:-"$HOME/bench/phase_current_serving_snapshot/$(date +%Y%m%d_%H%M%S)"}
|
||||
NPL=${NPL:-"8 32 128"}
|
||||
PTOK=${PTOK:-128}
|
||||
GEN=${GEN:-64}
|
||||
CTX=${CTX:-131072}
|
||||
PARALLEL=${PARALLEL:-128}
|
||||
BATCH=${BATCH:-2048}
|
||||
UBATCH=${UBATCH:-512}
|
||||
LLAMA_PORT=${LLAMA_PORT:-8098}
|
||||
LLAMA_READY_ATTEMPTS=${LLAMA_READY_ATTEMPTS:-240}
|
||||
VLLM_PORT=${VLLM_PORT:-8000}
|
||||
VLLM_BIN=${VLLM_BIN:-"$HOME/vllm-bench/bin/vllm"}
|
||||
VLLM_READY_ATTEMPTS=${VLLM_READY_ATTEMPTS:-600}
|
||||
VLLM_GPU_MEMORY_UTILIZATION=${VLLM_GPU_MEMORY_UTILIZATION:-0.85}
|
||||
VLLM_MAX_MODEL_LEN=${VLLM_MAX_MODEL_LEN:-4096}
|
||||
VLLM_MAX_NUM_SEQS=${VLLM_MAX_NUM_SEQS:-256}
|
||||
VLLM_TENSOR_PARALLEL_SIZE=${VLLM_TENSOR_PARALLEL_SIZE:-1}
|
||||
VLLM_EXTRA_ARGS=${VLLM_EXTRA_ARGS:-}
|
||||
SKIP_GATES=${SKIP_GATES:-0}
|
||||
DRY_RUN=${DRY_RUN:-0}
|
||||
MOE_MD5_EXPECTED=8cb0ce23777bf55f92f63d0292c756b0
|
||||
DENSE_MD5_EXPECTED=5951a5b4d624ce891e22ab5fca9bc439
|
||||
|
||||
LOCK_DIR="$HOME/gpu_bench_lock"
|
||||
OWNER="$LOCK_DIR/owner"
|
||||
SERVER_PID=""
|
||||
|
||||
log() {
|
||||
printf '[%s] %s\n' "$(date -Is)" "$*" | tee -a "$ART/run.log"
|
||||
}
|
||||
|
||||
require_path() {
|
||||
if [[ ! -e "$1" ]]; then
|
||||
echo "missing required path: $1" >&2
|
||||
exit 2
|
||||
fi
|
||||
}
|
||||
|
||||
preflight() {
|
||||
mkdir -p "$ART"
|
||||
local docker_count local_ai compute owner
|
||||
docker_count=$(docker ps -q | wc -l)
|
||||
local_ai=$(docker ps --format "{{.Names}}" | grep -c local-ai-worker || true)
|
||||
compute=$(nvidia-smi --query-compute-apps=pid --format=csv,noheader | sed '/^$/d' | wc -l)
|
||||
owner="FREE-no-lock-file"
|
||||
if [[ -f "$OWNER" ]]; then
|
||||
owner=$(cat "$OWNER")
|
||||
fi
|
||||
{
|
||||
echo "docker=$docker_count"
|
||||
echo "local_ai_worker=$local_ai"
|
||||
echo "compute=$compute"
|
||||
echo "$owner"
|
||||
} | tee "$ART/preflight.txt"
|
||||
[[ "$docker_count" == "0" ]]
|
||||
[[ "$local_ai" == "0" ]]
|
||||
[[ "$compute" == "0" ]]
|
||||
case "$owner" in
|
||||
FREE*|FREE-no-lock-file) ;;
|
||||
*) echo "GPU lock is busy: $owner" >&2; exit 3 ;;
|
||||
esac
|
||||
}
|
||||
|
||||
write_hardware_report() {
|
||||
local out="$ART/hardware.txt"
|
||||
local gpu_name hardware_class
|
||||
|
||||
gpu_name=$(nvidia-smi --query-gpu=name --format=csv,noheader 2>/dev/null | head -1 || true)
|
||||
hardware_class="unknown"
|
||||
case "$gpu_name" in
|
||||
*B200*|*B100*|*GB200*) hardware_class="datacenter_blackwell" ;;
|
||||
*H200*|*H100*) hardware_class="datacenter_other" ;;
|
||||
*GB10*|*"DGX Spark"*|*RTX*|*"PRO 6000"*) hardware_class="gb10_or_workstation_blackwell" ;;
|
||||
esac
|
||||
|
||||
{
|
||||
echo "nvidia_smi_L:"
|
||||
nvidia-smi -L || true
|
||||
echo
|
||||
echo "nvidia_smi_query:"
|
||||
if ! nvidia-smi --query-gpu=name,driver_version,memory.total,compute_cap --format=csv,noheader; then
|
||||
nvidia-smi --query-gpu=name,driver_version,memory.total --format=csv,noheader || true
|
||||
fi
|
||||
echo
|
||||
echo "gpu_name=$gpu_name"
|
||||
echo "hardware_class=$hardware_class"
|
||||
case "$hardware_class" in
|
||||
datacenter_blackwell)
|
||||
echo "parity_note=datacenter Blackwell hardware: full parity methodology can choose new levers"
|
||||
;;
|
||||
datacenter_other)
|
||||
echo "parity_note=datacenter non-Blackwell hardware: do not generalize GB10 parity decisions"
|
||||
;;
|
||||
gb10_or_workstation_blackwell)
|
||||
echo "parity_note=GB10/workstation Blackwell hardware: GB10 shortcut closures apply unless new evidence says otherwise"
|
||||
;;
|
||||
*)
|
||||
echo "parity_note=unknown hardware: classify before making parity claims"
|
||||
;;
|
||||
esac
|
||||
} > "$out"
|
||||
log "hardware report: $out"
|
||||
}
|
||||
|
||||
acquire_lock() {
|
||||
mkdir -p "$LOCK_DIR"
|
||||
echo "codex-current-serving-snapshot $(date +%s)" > "$OWNER"
|
||||
}
|
||||
|
||||
release_lock() {
|
||||
stop_server_pid
|
||||
pkill -9 -f "[l]lama-server.*--port $LLAMA_PORT" >/dev/null 2>&1 || true
|
||||
pkill -9 -u "$(id -u)" -f "[v]llm serve" >/dev/null 2>&1 || true
|
||||
mkdir -p "$LOCK_DIR"
|
||||
echo "FREE released-by-codex-current-serving-snapshot $(date +%s)" > "$OWNER"
|
||||
}
|
||||
|
||||
stop_server_pid() {
|
||||
if [[ -n "$SERVER_PID" ]]; then
|
||||
kill "$SERVER_PID" >/dev/null 2>&1 || true
|
||||
for _ in $(seq 1 30); do
|
||||
if ! kill -0 "$SERVER_PID" >/dev/null 2>&1; then
|
||||
break
|
||||
fi
|
||||
sleep 1
|
||||
done
|
||||
if kill -0 "$SERVER_PID" >/dev/null 2>&1; then
|
||||
kill -9 "$SERVER_PID" >/dev/null 2>&1 || true
|
||||
fi
|
||||
wait "$SERVER_PID" >/dev/null 2>&1 || true
|
||||
SERVER_PID=""
|
||||
fi
|
||||
}
|
||||
|
||||
wait_http() {
|
||||
local url="$1"
|
||||
local pattern="$2"
|
||||
local log_file="$3"
|
||||
local health="$4"
|
||||
local attempts="$5"
|
||||
for _ in $(seq 1 "$attempts"); do
|
||||
if curl --max-time 2 -fsS "$url" > "$health" 2>"$health.err" && grep -q "$pattern" "$health"; then
|
||||
return 0
|
||||
fi
|
||||
if [[ -n "$SERVER_PID" ]] && ! kill -0 "$SERVER_PID" >/dev/null 2>&1; then
|
||||
tail -120 "$log_file" >&2 || true
|
||||
return 1
|
||||
fi
|
||||
sleep 1
|
||||
done
|
||||
tail -120 "$log_file" >&2 || true
|
||||
return 1
|
||||
}
|
||||
|
||||
run_gate() {
|
||||
local name="$1"
|
||||
if [[ "$SKIP_GATES" == "1" ]]; then
|
||||
log "skipping $name inference gate"
|
||||
return
|
||||
fi
|
||||
log "running $name inference gate"
|
||||
ART="$ART/gate_$name" "$HOME/paged-inference-gates.sh" > "$ART/gate_$name.log" 2>&1
|
||||
cat "$ART/gate_$name.log" | tee -a "$ART/run.log"
|
||||
}
|
||||
|
||||
run_paged() {
|
||||
local arm_dir="$ART/paged"
|
||||
mkdir -p "$arm_dir"
|
||||
log "starting paged current-stack server"
|
||||
cd "$BIN"
|
||||
env LLAMA_KV_PAGED=1 LLAMA_MOE_FORCE_GRAPHS=1 GGML_NO_BACKTRACE=1 \
|
||||
./llama-server \
|
||||
-m "$MODEL" -ngl 99 -fa on -c "$CTX" -b "$BATCH" -ub "$UBATCH" \
|
||||
--parallel "$PARALLEL" --host 127.0.0.1 --port "$LLAMA_PORT" --no-webui \
|
||||
> "$arm_dir/server.log" 2>&1 &
|
||||
SERVER_PID=$!
|
||||
wait_http "http://127.0.0.1:$LLAMA_PORT/health" "ok" "$arm_dir/server.log" "$arm_dir/health.json" "$LLAMA_READY_ATTEMPTS"
|
||||
python3 "$H2H" --url "http://127.0.0.1:$LLAMA_PORT/v1/completions" \
|
||||
--model "$SERVED_MODEL_NAME" -n 8 --ptok "$PTOK" --gen 16 --nonce "warm_paged_$(date +%s)" --no-cache >/dev/null
|
||||
for n in $NPL; do
|
||||
log "paged n=$n"
|
||||
python3 "$H2H" --url "http://127.0.0.1:$LLAMA_PORT/v1/completions" \
|
||||
--model "$SERVED_MODEL_NAME" -n "$n" --ptok "$PTOK" --gen "$GEN" \
|
||||
--nonce "paged_${n}_$(date +%s)" --no-cache > "$arm_dir/n${n}.json"
|
||||
cat "$arm_dir/n${n}.json" | tee -a "$ART/run.log"
|
||||
done
|
||||
stop_server_pid
|
||||
sleep 3
|
||||
}
|
||||
|
||||
run_vllm() {
|
||||
local arm_dir="$ART/vllm"
|
||||
local extra_args=()
|
||||
mkdir -p "$arm_dir"
|
||||
export PATH="$(dirname "$VLLM_BIN"):$PATH"
|
||||
export VLLM_LOGGING_LEVEL=${VLLM_LOGGING_LEVEL:-INFO}
|
||||
export HF_HUB_OFFLINE=${HF_HUB_OFFLINE:-1}
|
||||
if [[ -n "$VLLM_EXTRA_ARGS" ]]; then
|
||||
read -r -a extra_args <<< "$VLLM_EXTRA_ARGS"
|
||||
fi
|
||||
log "starting vLLM server"
|
||||
nohup env \
|
||||
-u VLLM_MODEL -u VLLM_BIN -u VLLM_READY_ATTEMPTS \
|
||||
-u VLLM_GPU_MEMORY_UTILIZATION -u VLLM_MAX_MODEL_LEN -u VLLM_MAX_NUM_SEQS \
|
||||
-u VLLM_TENSOR_PARALLEL_SIZE -u VLLM_EXTRA_ARGS \
|
||||
"$VLLM_BIN" serve "$VLLM_MODEL" \
|
||||
--served-model-name "$SERVED_MODEL_NAME" --gpu-memory-utilization "$VLLM_GPU_MEMORY_UTILIZATION" --max-model-len "$VLLM_MAX_MODEL_LEN" \
|
||||
--max-num-seqs "$VLLM_MAX_NUM_SEQS" --host 127.0.0.1 --port "$VLLM_PORT" --tensor-parallel-size "$VLLM_TENSOR_PARALLEL_SIZE" \
|
||||
"${extra_args[@]}" \
|
||||
> "$arm_dir/server.log" 2>&1 &
|
||||
SERVER_PID=$!
|
||||
wait_http "http://127.0.0.1:$VLLM_PORT/v1/models" "$SERVED_MODEL_NAME" "$arm_dir/server.log" "$arm_dir/models.json" "$VLLM_READY_ATTEMPTS"
|
||||
python3 "$H2H" --url "http://127.0.0.1:$VLLM_PORT/v1/completions" \
|
||||
--model "$SERVED_MODEL_NAME" -n 8 --ptok "$PTOK" --gen 16 --nonce "warm_vllm_$(date +%s)" --no-cache >/dev/null
|
||||
for n in $NPL; do
|
||||
log "vllm n=$n"
|
||||
python3 "$H2H" --url "http://127.0.0.1:$VLLM_PORT/v1/completions" \
|
||||
--model "$SERVED_MODEL_NAME" -n "$n" --ptok "$PTOK" --gen "$GEN" \
|
||||
--nonce "vllm_${n}_$(date +%s)" --no-cache > "$arm_dir/n${n}.json"
|
||||
cat "$arm_dir/n${n}.json" | tee -a "$ART/run.log"
|
||||
done
|
||||
stop_server_pid
|
||||
pkill -9 -u "$(id -u)" -f "[v]llm serve" >/dev/null 2>&1 || true
|
||||
sleep 5
|
||||
}
|
||||
|
||||
write_summary() {
|
||||
python3 - "$ART" <<'PY' | tee "$ART/summary.tsv"
|
||||
import json
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
art = Path(sys.argv[1])
|
||||
rows = []
|
||||
for arm in ("paged", "vllm"):
|
||||
for path in sorted((art / arm).glob("n*.json")):
|
||||
data = json.loads(path.read_text())
|
||||
rows.append((arm, data["n"], data["agg_tps"], data["decode_agg_tps"],
|
||||
data["decode_perseq_tps"], data["prefill_tps"],
|
||||
data["ttft_mean_ms"], data["wall_s"]))
|
||||
|
||||
print("arm\tn\tagg_tps\tdecode_agg_tps\tdecode_perseq_tps\tprefill_tps\tttft_mean_ms\twall_s")
|
||||
for row in rows:
|
||||
print("\t".join(str(x) for x in row))
|
||||
|
||||
by_key = {(row[0], row[1]): row for row in rows}
|
||||
print("\nratio\tn\tpaged_decode_over_vllm\tpaged_perseq_over_vllm\tpaged_agg_over_vllm\tpaged_ttft_over_vllm")
|
||||
for n in sorted({row[1] for row in rows}):
|
||||
paged = by_key.get(("paged", n))
|
||||
vllm = by_key.get(("vllm", n))
|
||||
if not paged or not vllm:
|
||||
continue
|
||||
print(f"ratio\t{n}\t{paged[3]/vllm[3]:.4f}\t{paged[4]/vllm[4]:.4f}\t{paged[2]/vllm[2]:.4f}\t{paged[6]/vllm[6]:.4f}")
|
||||
PY
|
||||
}
|
||||
|
||||
write_gate_summary() {
|
||||
python3 - "$ART" "$MOE_MD5_EXPECTED" "$DENSE_MD5_EXPECTED" <<'PY' | tee "$ART/gate_summary.tsv"
|
||||
import re
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
art = Path(sys.argv[1])
|
||||
expected = {
|
||||
"moe": sys.argv[2],
|
||||
"dense": sys.argv[3],
|
||||
}
|
||||
ansi = re.compile(r"\x1b\[[0-9;]*m")
|
||||
bad = False
|
||||
|
||||
print("phase\tcheck\tstatus\tactual\texpected\tdetails")
|
||||
|
||||
for phase in ("pre", "post"):
|
||||
gate_dir = art / f"gate_{phase}"
|
||||
if not gate_dir.exists():
|
||||
print(f"{phase}\tall\tskipped\t\t\t{gate_dir} missing")
|
||||
continue
|
||||
|
||||
for name, want in expected.items():
|
||||
md5_path = gate_dir / f"{name}.md5"
|
||||
if not md5_path.exists():
|
||||
print(f"{phase}\t{name}_md5\tmissing\t\t{want}\t{md5_path} missing")
|
||||
bad = True
|
||||
continue
|
||||
got = md5_path.read_text().split()[0]
|
||||
status = "ok" if got == want else "mismatch"
|
||||
if status != "ok":
|
||||
bad = True
|
||||
print(f"{phase}\t{name}_md5\t{status}\t{got}\t{want}\t{md5_path}")
|
||||
|
||||
op_paths = sorted(gate_dir.glob("op_*.txt"))
|
||||
if not op_paths:
|
||||
print(f"{phase}\top\tmissing\t\t\tno op_*.txt files")
|
||||
bad = True
|
||||
continue
|
||||
|
||||
for path in op_paths:
|
||||
op = path.stem.removeprefix("op_")
|
||||
text = ansi.sub("", path.read_text(errors="replace"))
|
||||
passed = re.search(r"(\d+)/(\d+) tests passed", text)
|
||||
backend_ok = re.search(r"Backend CUDA0:\s+OK", text)
|
||||
if passed:
|
||||
actual = f"{passed.group(1)}/{passed.group(2)}"
|
||||
status = "ok" if passed.group(1) == passed.group(2) and backend_ok else "fail"
|
||||
else:
|
||||
actual = ""
|
||||
status = "missing"
|
||||
if status != "ok":
|
||||
bad = True
|
||||
print(f"{phase}\top_{op}\t{status}\t{actual}\tall\t{path}")
|
||||
|
||||
if bad:
|
||||
sys.exit(6)
|
||||
PY
|
||||
}
|
||||
|
||||
if [[ -n "$SUMMARY_GATES_ART" ]]; then
|
||||
ART="$SUMMARY_GATES_ART"
|
||||
require_path "$ART"
|
||||
write_gate_summary
|
||||
exit 0
|
||||
fi
|
||||
|
||||
require_path "$SRC"
|
||||
require_path "$BIN/llama-server"
|
||||
require_path "$BIN/llama-completion"
|
||||
require_path "$BIN/test-backend-ops"
|
||||
require_path "$MODEL"
|
||||
require_path "$VLLM_MODEL"
|
||||
require_path "$H2H"
|
||||
require_path "$VLLM_BIN"
|
||||
require_path "$HOME/paged-inference-gates.sh"
|
||||
|
||||
preflight
|
||||
write_hardware_report
|
||||
log "artifact=$ART"
|
||||
log "source=$(git -C "$SRC" log --oneline -1)"
|
||||
|
||||
if [[ "$DRY_RUN" == "1" ]]; then
|
||||
log "dry run only; commands validated"
|
||||
log "would build: cmake --build $BUILD_DIR --target llama-server llama-completion test-backend-ops -j8"
|
||||
log "served model: SERVED_MODEL_NAME=$SERVED_MODEL_NAME"
|
||||
log "readiness: LLAMA_READY_ATTEMPTS=$LLAMA_READY_ATTEMPTS VLLM_READY_ATTEMPTS=$VLLM_READY_ATTEMPTS"
|
||||
log "would run paged NPL=[$NPL] PTOK=$PTOK GEN=$GEN"
|
||||
log "would run vLLM NPL=[$NPL] PTOK=$PTOK GEN=$GEN"
|
||||
log "vLLM config: VLLM_GPU_MEMORY_UTILIZATION=$VLLM_GPU_MEMORY_UTILIZATION VLLM_MAX_MODEL_LEN=$VLLM_MAX_MODEL_LEN VLLM_MAX_NUM_SEQS=$VLLM_MAX_NUM_SEQS VLLM_TENSOR_PARALLEL_SIZE=$VLLM_TENSOR_PARALLEL_SIZE VLLM_EXTRA_ARGS=[$VLLM_EXTRA_ARGS]"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
log "building llama-server, llama-completion, and test-backend-ops"
|
||||
cmake --build "$BUILD_DIR" --target llama-server llama-completion test-backend-ops -j 8 \
|
||||
> "$ART/build.log" 2>&1
|
||||
|
||||
run_gate pre
|
||||
acquire_lock
|
||||
trap release_lock EXIT
|
||||
run_paged
|
||||
run_vllm
|
||||
release_lock
|
||||
trap - EXIT
|
||||
run_gate post
|
||||
write_gate_summary
|
||||
write_summary
|
||||
log "artifacts: $ART"
|
||||
@@ -1,136 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
if [[ "${1:-}" == "-h" || "${1:-}" == "--help" ]]; then
|
||||
cat <<'EOF'
|
||||
Usage: paged-inference-gates.sh
|
||||
|
||||
Run the LocalAI paged llama.cpp inference safety gates on a DGX checkout.
|
||||
|
||||
Environment:
|
||||
BIN llama.cpp build bin dir (default: ~/llama-phase6-source/build-cuda/bin)
|
||||
MOE MoE GGUF path (default: ~/bench/q36-35b-a3b-nvfp4.gguf)
|
||||
DENSE Dense GGUF path (default: ~/bench/q36-27b-nvfp4.gguf)
|
||||
ART artifact dir (default: ~/bench/paged_inference_gates/<timestamp>)
|
||||
OPS comma-separated test-backend-ops filters (default: MUL_MAT,MUL_MAT_ID)
|
||||
EXTRA_ENV extra env assignments for completion gates, e.g. "GDN_TC=5"
|
||||
|
||||
Expected md5:
|
||||
MoE paged: 8cb0ce23777bf55f92f63d0292c756b0
|
||||
Dense paged: 5951a5b4d624ce891e22ab5fca9bc439
|
||||
EOF
|
||||
exit 0
|
||||
fi
|
||||
|
||||
MOE_MD5_EXPECTED=8cb0ce23777bf55f92f63d0292c756b0
|
||||
DENSE_MD5_EXPECTED=5951a5b4d624ce891e22ab5fca9bc439
|
||||
|
||||
BIN=${BIN:-"$HOME/llama-phase6-source/build-cuda/bin"}
|
||||
MOE=${MOE:-"$HOME/bench/q36-35b-a3b-nvfp4.gguf"}
|
||||
DENSE=${DENSE:-"$HOME/bench/q36-27b-nvfp4.gguf"}
|
||||
OPS=${OPS:-MUL_MAT,MUL_MAT_ID}
|
||||
ART=${ART:-"$HOME/bench/paged_inference_gates/$(date +%Y%m%d_%H%M%S)"}
|
||||
EXTRA_ENV=${EXTRA_ENV:-}
|
||||
|
||||
require_file() {
|
||||
if [[ ! -e "$1" ]]; then
|
||||
echo "missing required path: $1" >&2
|
||||
exit 2
|
||||
fi
|
||||
}
|
||||
|
||||
check_idle() {
|
||||
if command -v docker >/dev/null 2>&1; then
|
||||
local docker_count
|
||||
docker_count=$(docker ps -q | wc -l)
|
||||
if [[ "$docker_count" != "0" ]]; then
|
||||
echo "docker containers are running: $docker_count" >&2
|
||||
docker ps >&2
|
||||
exit 3
|
||||
fi
|
||||
|
||||
local local_ai_worker
|
||||
local_ai_worker=$(docker ps --format "{{.Names}}" | grep -c local-ai-worker || true)
|
||||
if [[ "$local_ai_worker" != "0" ]]; then
|
||||
echo "local-ai-worker container is running" >&2
|
||||
exit 3
|
||||
fi
|
||||
fi
|
||||
|
||||
if command -v nvidia-smi >/dev/null 2>&1; then
|
||||
local compute_count
|
||||
compute_count=$(nvidia-smi --query-compute-apps=pid --format=csv,noheader | sed "/^$/d" | wc -l)
|
||||
if [[ "$compute_count" != "0" ]]; then
|
||||
echo "GPU compute processes are already running: $compute_count" >&2
|
||||
nvidia-smi >&2
|
||||
exit 3
|
||||
fi
|
||||
fi
|
||||
|
||||
local owner_file="$HOME/gpu_bench_lock/owner"
|
||||
if [[ -f "$owner_file" ]]; then
|
||||
local owner
|
||||
owner=$(cat "$owner_file")
|
||||
if [[ -n "$owner" && "$owner" != FREE* ]]; then
|
||||
echo "GPU lock is owned: $owner" >&2
|
||||
exit 3
|
||||
fi
|
||||
fi
|
||||
}
|
||||
|
||||
run_completion_gate() {
|
||||
local name=$1
|
||||
local model=$2
|
||||
local expected=$3
|
||||
local out="$ART/${name}.txt"
|
||||
local err="$ART/${name}.err"
|
||||
local md5_file="$ART/${name}.md5"
|
||||
|
||||
env LLAMA_KV_PAGED=1 LLAMA_MOE_FORCE_GRAPHS=1 GGML_NO_BACKTRACE=1 $EXTRA_ENV \
|
||||
"$BIN/llama-completion" -m "$model" -ngl 99 -fa on -c 4096 \
|
||||
--temp 0 --seed 1 -n 48 -p "The capital of France is" \
|
||||
</dev/null >"$out" 2>"$err"
|
||||
|
||||
md5sum "$out" >"$md5_file"
|
||||
local actual
|
||||
actual=$(awk '{print $1}' "$md5_file")
|
||||
if [[ "$actual" != "$expected" ]]; then
|
||||
echo "$name md5 mismatch: got $actual expected $expected" >&2
|
||||
echo "artifacts: $ART" >&2
|
||||
exit 4
|
||||
fi
|
||||
echo "$name md5 OK: $actual"
|
||||
}
|
||||
|
||||
run_op_gate() {
|
||||
local op=$1
|
||||
local out="$ART/op_${op}.txt"
|
||||
"$BIN/test-backend-ops" test -b CUDA0 -o "$op" -j 1 >"$out" 2>&1
|
||||
if ! grep -q "Backend CUDA0: .*OK" "$out"; then
|
||||
echo "$op gate failed" >&2
|
||||
tail -80 "$out" >&2
|
||||
echo "artifacts: $ART" >&2
|
||||
exit 5
|
||||
fi
|
||||
grep -E "[0-9]+/[0-9]+ tests passed|Backend CUDA0" "$out" | tail -2
|
||||
}
|
||||
|
||||
mkdir -p "$ART"
|
||||
require_file "$BIN/llama-completion"
|
||||
require_file "$BIN/test-backend-ops"
|
||||
require_file "$MOE"
|
||||
require_file "$DENSE"
|
||||
check_idle
|
||||
|
||||
run_completion_gate moe "$MOE" "$MOE_MD5_EXPECTED"
|
||||
run_completion_gate dense "$DENSE" "$DENSE_MD5_EXPECTED"
|
||||
|
||||
IFS=',' read -r -a op_list <<<"$OPS"
|
||||
for op in "${op_list[@]}"; do
|
||||
op=${op//[[:space:]]/}
|
||||
[[ -n "$op" ]] || continue
|
||||
run_op_gate "$op"
|
||||
done
|
||||
|
||||
echo "paged inference gates OK"
|
||||
echo "artifacts: $ART"
|
||||
@@ -1,200 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
usage() {
|
||||
cat <<'EOF'
|
||||
Usage: paged-mtp-serving-bench.sh
|
||||
|
||||
Runs a direct llama-server serving A/B on DGX:
|
||||
baseline: no speculative decoding
|
||||
mtp: --spec-type draft-mtp
|
||||
|
||||
Environment overrides:
|
||||
SRC llama.cpp source dir (default: ~/llama-phase6-source)
|
||||
BIN binary dir (default: $SRC/build-cuda/bin)
|
||||
MODEL MoE GGUF path (default: ~/bench/q36-35b-a3b-nvfp4.gguf)
|
||||
ART artifact dir (default: ~/bench/phase15_mtp_serving/<timestamp>)
|
||||
PORT server port (default: 8097)
|
||||
NPL comma/space list of concurrency values (default: "8 32 128")
|
||||
PTOK prompt filler words for h2h_cli3.py (default: 128)
|
||||
GEN max generated tokens (default: 128)
|
||||
CTX server context (default: 131072)
|
||||
PARALLEL server parallel slots (default: 128)
|
||||
BATCH server logical batch size (default: 2048)
|
||||
UBATCH server physical batch size (default: 512)
|
||||
SKIP_GATES=1 to skip pre/post paged inference gates
|
||||
EOF
|
||||
}
|
||||
|
||||
if [[ "${1:-}" == "-h" || "${1:-}" == "--help" ]]; then
|
||||
usage
|
||||
exit 0
|
||||
fi
|
||||
|
||||
SRC=${SRC:-"$HOME/llama-phase6-source"}
|
||||
BIN=${BIN:-"$SRC/build-cuda/bin"}
|
||||
MODEL=${MODEL:-"$HOME/bench/q36-35b-a3b-nvfp4.gguf"}
|
||||
ART=${ART:-"$HOME/bench/phase15_mtp_serving/$(date +%Y%m%d_%H%M%S)"}
|
||||
PORT=${PORT:-8097}
|
||||
NPL=${NPL:-"8 32 128"}
|
||||
PTOK=${PTOK:-128}
|
||||
GEN=${GEN:-128}
|
||||
CTX=${CTX:-131072}
|
||||
PARALLEL=${PARALLEL:-128}
|
||||
BATCH=${BATCH:-2048}
|
||||
UBATCH=${UBATCH:-512}
|
||||
SKIP_GATES=${SKIP_GATES:-0}
|
||||
|
||||
LOCK_DIR="$HOME/gpu_bench_lock"
|
||||
OWNER="$LOCK_DIR/owner"
|
||||
SERVER_PID=""
|
||||
|
||||
log() {
|
||||
printf '[%s] %s\n' "$(date -Is)" "$*" | tee -a "$ART/run.log"
|
||||
}
|
||||
|
||||
preflight() {
|
||||
mkdir -p "$ART"
|
||||
local docker_count local_ai compute owner
|
||||
docker_count=$(docker ps -q | wc -l)
|
||||
local_ai=$(docker ps --format "{{.Names}}" | grep -c local-ai-worker || true)
|
||||
compute=$(nvidia-smi --query-compute-apps=pid --format=csv,noheader | sed '/^$/d' | wc -l)
|
||||
owner="FREE-no-lock-file"
|
||||
if [[ -f "$OWNER" ]]; then
|
||||
owner=$(cat "$OWNER")
|
||||
fi
|
||||
{
|
||||
echo "docker=$docker_count"
|
||||
echo "local_ai_worker=$local_ai"
|
||||
echo "compute=$compute"
|
||||
echo "$owner"
|
||||
} | tee "$ART/preflight.txt"
|
||||
[[ "$docker_count" == "0" ]]
|
||||
[[ "$local_ai" == "0" ]]
|
||||
[[ "$compute" == "0" ]]
|
||||
case "$owner" in
|
||||
FREE*|FREE-no-lock-file) ;;
|
||||
*) echo "GPU lock is busy: $owner" >&2; exit 2 ;;
|
||||
esac
|
||||
}
|
||||
|
||||
acquire_lock() {
|
||||
mkdir -p "$LOCK_DIR"
|
||||
echo "codex-phase15-mtp-serving-bench $(date +%s)" > "$OWNER"
|
||||
}
|
||||
|
||||
release_lock() {
|
||||
if [[ -n "$SERVER_PID" ]]; then
|
||||
kill "$SERVER_PID" >/dev/null 2>&1 || true
|
||||
wait "$SERVER_PID" >/dev/null 2>&1 || true
|
||||
SERVER_PID=""
|
||||
fi
|
||||
mkdir -p "$LOCK_DIR"
|
||||
echo "FREE released-by-codex-phase15-mtp-serving-bench $(date +%s)" > "$OWNER"
|
||||
}
|
||||
|
||||
wait_server() {
|
||||
local health="$1"
|
||||
for _ in $(seq 1 180); do
|
||||
if curl -fsS "http://127.0.0.1:$PORT/health" > "$health" 2>"$health.err"; then
|
||||
return 0
|
||||
fi
|
||||
if ! kill -0 "$SERVER_PID" 2>/dev/null; then
|
||||
return 1
|
||||
fi
|
||||
sleep 1
|
||||
done
|
||||
return 1
|
||||
}
|
||||
|
||||
stop_server() {
|
||||
if [[ -n "$SERVER_PID" ]]; then
|
||||
kill "$SERVER_PID" >/dev/null 2>&1 || true
|
||||
wait "$SERVER_PID" >/dev/null 2>&1 || true
|
||||
SERVER_PID=""
|
||||
fi
|
||||
}
|
||||
|
||||
run_gate() {
|
||||
local name="$1"
|
||||
if [[ "$SKIP_GATES" == "1" ]]; then
|
||||
log "skipping $name inference gate"
|
||||
return
|
||||
fi
|
||||
log "running $name inference gate"
|
||||
ART="$ART/gate_$name" "$HOME/paged-inference-gates.sh" > "$ART/gate_$name.log" 2>&1
|
||||
cat "$ART/gate_$name.log" | tee -a "$ART/run.log"
|
||||
}
|
||||
|
||||
run_arm() {
|
||||
local arm="$1"
|
||||
shift
|
||||
local arm_dir="$ART/$arm"
|
||||
mkdir -p "$arm_dir"
|
||||
log "starting $arm server"
|
||||
cd "$BIN"
|
||||
env LLAMA_KV_PAGED=1 LLAMA_MOE_FORCE_GRAPHS=1 GGML_NO_BACKTRACE=1 \
|
||||
./llama-server \
|
||||
-m "$MODEL" -ngl 99 -fa on -c "$CTX" -b "$BATCH" -ub "$UBATCH" \
|
||||
--parallel "$PARALLEL" --host 127.0.0.1 --port "$PORT" --no-webui "$@" \
|
||||
> "$arm_dir/server.log" 2>&1 &
|
||||
SERVER_PID=$!
|
||||
if ! wait_server "$arm_dir/health.json"; then
|
||||
tail -120 "$arm_dir/server.log" >&2 || true
|
||||
exit 3
|
||||
fi
|
||||
|
||||
for n in $NPL; do
|
||||
log "running $arm n=$n"
|
||||
python3 "$HOME/bench/h2h_cli3.py" \
|
||||
--url "http://127.0.0.1:$PORT/v1/completions" \
|
||||
--model m -n "$n" --ptok "$PTOK" --gen "$GEN" \
|
||||
--nonce "${arm}_${n}_$(date +%s)" --no-cache \
|
||||
> "$arm_dir/n${n}.json"
|
||||
cat "$arm_dir/n${n}.json" | tee -a "$ART/run.log"
|
||||
done
|
||||
|
||||
grep -E "draft acceptance|statistics[[:space:]]+draft-mtp|speculative decoding context|bounded partial|backend sampling|common_speculative_impl_draft_mtp" \
|
||||
"$arm_dir/server.log" > "$arm_dir/spec_lines.txt" || true
|
||||
stop_server
|
||||
}
|
||||
|
||||
preflight
|
||||
|
||||
log "building llama-server and test-backend-ops"
|
||||
cmake --build "$SRC/build-cuda" --target llama-server test-backend-ops llama-completion -j 8 \
|
||||
> "$ART/build.log" 2>&1
|
||||
|
||||
if [[ ! -x "$HOME/paged-inference-gates.sh" ]]; then
|
||||
echo "missing $HOME/paged-inference-gates.sh; copy paged-inference-gates.sh there first" >&2
|
||||
exit 4
|
||||
fi
|
||||
|
||||
run_gate pre
|
||||
acquire_lock
|
||||
trap release_lock EXIT
|
||||
run_arm baseline
|
||||
run_arm mtp --spec-type draft-mtp --spec-draft-n-max 3 --no-spec-draft-backend-sampling
|
||||
release_lock
|
||||
trap - EXIT
|
||||
run_gate post
|
||||
|
||||
python3 - "$ART" <<'PY' | tee "$ART/summary.tsv"
|
||||
import json
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
art = Path(sys.argv[1])
|
||||
rows = []
|
||||
for arm in ("baseline", "mtp"):
|
||||
for path in sorted((art / arm).glob("n*.json")):
|
||||
data = json.loads(path.read_text())
|
||||
rows.append((arm, data["n"], data["gen_total"], data["agg_tps"],
|
||||
data["decode_agg_tps"], data["decode_perseq_tps"],
|
||||
data["ttft_mean_ms"], data["wall_s"]))
|
||||
print("arm\tn\tgen_total\tagg_tps\tdecode_agg_tps\tdecode_perseq_tps\tttft_mean_ms\twall_s")
|
||||
for row in rows:
|
||||
print("\t".join(str(x) for x in row))
|
||||
PY
|
||||
|
||||
log "artifacts: $ART"
|
||||
@@ -1,448 +0,0 @@
|
||||
From bef64835d444a44ed8391bc395cdab38164229d5 Mon Sep 17 00:00:00 2001
|
||||
From: Ettore Di Giacinto <mudler@localai.io>
|
||||
Date: Fri, 19 Jun 2026 22:54:49 +0000
|
||||
Subject: [PATCH] vendor paged kv manager
|
||||
|
||||
vLLM-parity host-side KV block manager (FreeBlockQueue, BlockPool,
|
||||
PagedKVManager, chained-hash prefix cache). Pure C++17, no behavior change -
|
||||
nothing uses it yet; wired in by later patches in the series.
|
||||
---
|
||||
src/CMakeLists.txt | 1 +
|
||||
src/paged-kv-manager.cpp | 296 +++++++++++++++++++++++++++++++++++++++
|
||||
src/paged-kv-manager.h | 108 ++++++++++++++
|
||||
3 files changed, 405 insertions(+)
|
||||
create mode 100644 src/paged-kv-manager.cpp
|
||||
create mode 100644 src/paged-kv-manager.h
|
||||
|
||||
diff --git a/src/CMakeLists.txt b/src/CMakeLists.txt
|
||||
index d15ccfd99..a030940b8 100644
|
||||
--- a/src/CMakeLists.txt
|
||||
+++ b/src/CMakeLists.txt
|
||||
@@ -24,6 +24,7 @@ add_library(llama
|
||||
llama-io.cpp
|
||||
llama-kv-cache.cpp
|
||||
llama-kv-cache-iswa.cpp
|
||||
+ paged-kv-manager.cpp
|
||||
llama-kv-cache-dsa.cpp
|
||||
llama-memory.cpp
|
||||
llama-memory-hybrid.cpp
|
||||
diff --git a/src/paged-kv-manager.cpp b/src/paged-kv-manager.cpp
|
||||
new file mode 100644
|
||||
index 000000000..ca0dcd83a
|
||||
--- /dev/null
|
||||
+++ b/src/paged-kv-manager.cpp
|
||||
@@ -0,0 +1,296 @@
|
||||
+#include "paged-kv-manager.h"
|
||||
+#include <cassert>
|
||||
+#include <stdexcept>
|
||||
+
|
||||
+namespace paged {
|
||||
+
|
||||
+// ---------------------------------------------------------------------------
|
||||
+// FreeBlockQueue (port of kv_cache_utils.py FreeKVCacheBlockQueue)
|
||||
+// ---------------------------------------------------------------------------
|
||||
+
|
||||
+FreeBlockQueue::FreeBlockQueue(const std::vector<KVCacheBlock*>& blocks) {
|
||||
+ num_free_blocks = blocks.size();
|
||||
+ for (size_t i = 0; i < blocks.size(); ++i) {
|
||||
+ if (i > 0) blocks[i]->prev_free = blocks[i - 1];
|
||||
+ if (i + 1 < blocks.size()) blocks[i]->next_free = blocks[i + 1];
|
||||
+ }
|
||||
+ if (!blocks.empty()) {
|
||||
+ fake_head.next_free = blocks.front();
|
||||
+ blocks.front()->prev_free = &fake_head;
|
||||
+ fake_tail.prev_free = blocks.back();
|
||||
+ blocks.back()->next_free = &fake_tail;
|
||||
+ } else {
|
||||
+ fake_head.next_free = &fake_tail;
|
||||
+ fake_tail.prev_free = &fake_head;
|
||||
+ }
|
||||
+}
|
||||
+
|
||||
+KVCacheBlock* FreeBlockQueue::popleft() {
|
||||
+ KVCacheBlock* first = fake_head.next_free;
|
||||
+ if (first == &fake_tail || first == nullptr) {
|
||||
+ assert(num_free_blocks == 0);
|
||||
+ throw std::runtime_error("No free blocks available");
|
||||
+ }
|
||||
+ fake_head.next_free = first->next_free;
|
||||
+ first->next_free->prev_free = &fake_head;
|
||||
+ first->prev_free = first->next_free = nullptr;
|
||||
+ num_free_blocks--;
|
||||
+ return first;
|
||||
+}
|
||||
+
|
||||
+std::vector<KVCacheBlock*> FreeBlockQueue::popleft_n(size_t n) {
|
||||
+ std::vector<KVCacheBlock*> ret;
|
||||
+ if (n == 0) return ret;
|
||||
+ assert(num_free_blocks >= n);
|
||||
+ num_free_blocks -= n;
|
||||
+ KVCacheBlock* curr = fake_head.next_free;
|
||||
+ ret.reserve(n);
|
||||
+ for (size_t i = 0; i < n; ++i) {
|
||||
+ assert(curr != nullptr);
|
||||
+ ret.push_back(curr);
|
||||
+ KVCacheBlock* last = curr;
|
||||
+ curr = curr->next_free;
|
||||
+ last->prev_free = last->next_free = nullptr;
|
||||
+ }
|
||||
+ if (curr != nullptr) {
|
||||
+ fake_head.next_free = curr;
|
||||
+ curr->prev_free = &fake_head;
|
||||
+ }
|
||||
+ return ret;
|
||||
+}
|
||||
+
|
||||
+void FreeBlockQueue::remove(KVCacheBlock* block) {
|
||||
+ if (!block->prev_free || !block->next_free)
|
||||
+ throw std::runtime_error("remove() called on an invalid block");
|
||||
+ block->prev_free->next_free = block->next_free;
|
||||
+ block->next_free->prev_free = block->prev_free;
|
||||
+ block->prev_free = block->next_free = nullptr;
|
||||
+ num_free_blocks--;
|
||||
+}
|
||||
+
|
||||
+void FreeBlockQueue::append(KVCacheBlock* block) {
|
||||
+ KVCacheBlock* last = fake_tail.prev_free;
|
||||
+ last->next_free = block;
|
||||
+ block->prev_free = last;
|
||||
+ block->next_free = &fake_tail;
|
||||
+ fake_tail.prev_free = block;
|
||||
+ num_free_blocks++;
|
||||
+}
|
||||
+
|
||||
+void FreeBlockQueue::append_n(const std::vector<KVCacheBlock*>& blocks) {
|
||||
+ if (blocks.empty()) return;
|
||||
+ KVCacheBlock* last = fake_tail.prev_free;
|
||||
+ for (KVCacheBlock* b : blocks) {
|
||||
+ b->prev_free = last;
|
||||
+ last->next_free = b;
|
||||
+ last = b;
|
||||
+ }
|
||||
+ last->next_free = &fake_tail;
|
||||
+ fake_tail.prev_free = last;
|
||||
+ num_free_blocks += blocks.size();
|
||||
+}
|
||||
+
|
||||
+void FreeBlockQueue::prepend_n(const std::vector<KVCacheBlock*>& blocks) {
|
||||
+ if (blocks.empty()) return;
|
||||
+ KVCacheBlock* first = fake_head.next_free;
|
||||
+ KVCacheBlock* prev = &fake_head;
|
||||
+ for (KVCacheBlock* b : blocks) {
|
||||
+ b->prev_free = prev;
|
||||
+ prev->next_free = b;
|
||||
+ prev = b;
|
||||
+ }
|
||||
+ prev->next_free = first;
|
||||
+ first->prev_free = prev;
|
||||
+ num_free_blocks += blocks.size();
|
||||
+}
|
||||
+
|
||||
+std::vector<KVCacheBlock*> FreeBlockQueue::get_all_free_blocks() const {
|
||||
+ std::vector<KVCacheBlock*> ret;
|
||||
+ const KVCacheBlock* curr = fake_head.next_free;
|
||||
+ while (curr && curr->next_free != nullptr) {
|
||||
+ ret.push_back(const_cast<KVCacheBlock*>(curr));
|
||||
+ curr = curr->next_free;
|
||||
+ }
|
||||
+ return ret;
|
||||
+}
|
||||
+
|
||||
+// ---------------------------------------------------------------------------
|
||||
+// BlockPool (port of block_pool.py)
|
||||
+// ---------------------------------------------------------------------------
|
||||
+
|
||||
+static std::vector<KVCacheBlock*> make_ptrs(std::vector<KVCacheBlock>& v) {
|
||||
+ std::vector<KVCacheBlock*> p;
|
||||
+ p.reserve(v.size());
|
||||
+ for (auto& b : v) p.push_back(&b);
|
||||
+ return p;
|
||||
+}
|
||||
+
|
||||
+static std::vector<KVCacheBlock> make_block_vec(int32_t num_blocks) {
|
||||
+ std::vector<KVCacheBlock> v;
|
||||
+ v.reserve(num_blocks);
|
||||
+ for (int32_t i = 0; i < num_blocks; ++i) v.emplace_back(i);
|
||||
+ return v;
|
||||
+}
|
||||
+
|
||||
+BlockPool::BlockPool(int32_t num_blocks, bool enable_caching)
|
||||
+ : enable_caching_(enable_caching),
|
||||
+ blocks_(make_block_vec(num_blocks)),
|
||||
+ ptrs_(make_ptrs(blocks_)),
|
||||
+ free_queue_(ptrs_) {
|
||||
+ // vLLM reserves block_id 0 as the null block (never cached).
|
||||
+ null_block = free_queue_.popleft();
|
||||
+ null_block->is_null = true;
|
||||
+}
|
||||
+
|
||||
+bool BlockPool::maybe_evict_cached_block(KVCacheBlock* block) {
|
||||
+ if (!block->has_hash) return false;
|
||||
+ auto it = cached_block_hash_to_block_.find(block->block_hash);
|
||||
+ if (it == cached_block_hash_to_block_.end() || it->second != block) return false;
|
||||
+ cached_block_hash_to_block_.erase(it);
|
||||
+ block->reset_hash();
|
||||
+ return true;
|
||||
+}
|
||||
+
|
||||
+std::vector<KVCacheBlock*> BlockPool::get_new_blocks(size_t n) {
|
||||
+ if (n > get_num_free_blocks())
|
||||
+ throw std::runtime_error("Cannot get free blocks from pool");
|
||||
+ auto ret = free_queue_.popleft_n(n);
|
||||
+ for (KVCacheBlock* b : ret) {
|
||||
+ if (enable_caching_) maybe_evict_cached_block(b);
|
||||
+ assert(b->ref_cnt == 0);
|
||||
+ b->ref_cnt += 1;
|
||||
+ }
|
||||
+ return ret;
|
||||
+}
|
||||
+
|
||||
+KVCacheBlock* BlockPool::get_cached_block(uint64_t block_hash) {
|
||||
+ auto it = cached_block_hash_to_block_.find(block_hash);
|
||||
+ return it == cached_block_hash_to_block_.end() ? nullptr : it->second;
|
||||
+}
|
||||
+
|
||||
+void BlockPool::touch(const std::vector<KVCacheBlock*>& blocks) {
|
||||
+ for (KVCacheBlock* b : blocks) {
|
||||
+ // ref_cnt==0 means the block is a free-list eviction candidate; pull it out.
|
||||
+ if (b->ref_cnt == 0 && !b->is_null) free_queue_.remove(b);
|
||||
+ b->ref_cnt += 1;
|
||||
+ }
|
||||
+}
|
||||
+
|
||||
+void BlockPool::free_blocks(const std::vector<KVCacheBlock*>& ordered_blocks) {
|
||||
+ std::vector<KVCacheBlock*> without_hash, with_hash;
|
||||
+ for (KVCacheBlock* b : ordered_blocks) {
|
||||
+ if (b->is_null) continue;
|
||||
+ b->ref_cnt -= 1;
|
||||
+ if (b->ref_cnt == 0) (b->has_hash ? with_hash : without_hash).push_back(b);
|
||||
+ }
|
||||
+ free_queue_.prepend_n(without_hash); // un-hashed: evicted first (front)
|
||||
+ free_queue_.append_n(with_hash); // hashed: kept warm (tail)
|
||||
+}
|
||||
+
|
||||
+void BlockPool::cache_full_blocks(const std::vector<KVCacheBlock*>& req_blocks,
|
||||
+ size_t num_cached_blocks, size_t num_full_blocks,
|
||||
+ const std::vector<uint64_t>& block_hashes) {
|
||||
+ for (size_t i = num_cached_blocks; i < num_full_blocks; ++i) {
|
||||
+ KVCacheBlock* blk = req_blocks[i];
|
||||
+ if (blk->has_hash) continue;
|
||||
+ blk->has_hash = true;
|
||||
+ blk->block_hash = block_hashes[i];
|
||||
+ cached_block_hash_to_block_[blk->block_hash] = blk;
|
||||
+ }
|
||||
+}
|
||||
+
|
||||
+// ---------------------------------------------------------------------------
|
||||
+// PagedKVManager (port of SingleTypeKVCacheManager / FullAttentionManager)
|
||||
+// ---------------------------------------------------------------------------
|
||||
+
|
||||
+static inline size_t cdiv(size_t a, size_t b) { return (a + b - 1) / b; }
|
||||
+
|
||||
+PagedKVManager::PagedKVManager(int32_t num_blocks, int block_size, bool enable_caching)
|
||||
+ : block_size_(block_size), pool_(num_blocks, enable_caching) {}
|
||||
+
|
||||
+bool PagedKVManager::allocate(int seq_id, size_t total_tokens) {
|
||||
+ auto& req = req_to_blocks_[seq_id];
|
||||
+ size_t need = cdiv(total_tokens, block_size_);
|
||||
+ if (need <= req.size()) return true;
|
||||
+ size_t add = need - req.size();
|
||||
+ if (add > pool_.get_num_free_blocks()) return false; // OOM
|
||||
+ auto nb = pool_.get_new_blocks(add);
|
||||
+ req.insert(req.end(), nb.begin(), nb.end());
|
||||
+ return true;
|
||||
+}
|
||||
+
|
||||
+std::vector<int32_t> PagedKVManager::block_table(int seq_id) const {
|
||||
+ std::vector<int32_t> bt;
|
||||
+ auto it = req_to_blocks_.find(seq_id);
|
||||
+ if (it == req_to_blocks_.end()) return bt;
|
||||
+ bt.reserve(it->second.size());
|
||||
+ for (KVCacheBlock* b : it->second) bt.push_back(b->block_id);
|
||||
+ return bt;
|
||||
+}
|
||||
+
|
||||
+int64_t PagedKVManager::slot(int seq_id, int pos) const {
|
||||
+ const auto& req = req_to_blocks_.at(seq_id);
|
||||
+ int32_t phys = req[pos / block_size_]->block_id;
|
||||
+ return (int64_t)phys * block_size_ + (pos % block_size_);
|
||||
+}
|
||||
+
|
||||
+std::vector<int64_t> PagedKVManager::slot_mapping(int seq_id, const std::vector<int>& positions) const {
|
||||
+ std::vector<int64_t> sm;
|
||||
+ sm.reserve(positions.size());
|
||||
+ for (int p : positions) sm.push_back(slot(seq_id, p));
|
||||
+ return sm;
|
||||
+}
|
||||
+
|
||||
+void PagedKVManager::free(int seq_id) {
|
||||
+ auto it = req_to_blocks_.find(seq_id);
|
||||
+ if (it == req_to_blocks_.end()) return;
|
||||
+ // Free in reverse so the tail of the block chain is evicted first (vLLM order).
|
||||
+ std::vector<KVCacheBlock*> ordered(it->second.rbegin(), it->second.rend());
|
||||
+ pool_.free_blocks(ordered);
|
||||
+ req_to_blocks_.erase(it);
|
||||
+}
|
||||
+
|
||||
+// FNV-1a chained block hash. Deterministic and prefix-sensitive; folds the parent
|
||||
+// hash into the seed so each block hash transitively encodes its whole prefix
|
||||
+// (behavioral parity with vLLM hash_block_tokens chaining; vLLM uses sha256 bytes).
|
||||
+uint64_t PagedKVManager::hash_block(uint64_t parent_hash, const std::vector<int>& token_ids) {
|
||||
+ uint64_t h = 1469598103934665603ull ^ parent_hash;
|
||||
+ for (int t : token_ids) {
|
||||
+ h ^= (uint64_t)(uint32_t)t;
|
||||
+ h *= 1099511628211ull;
|
||||
+ }
|
||||
+ if (h == 0) h = 0x9e3779b97f4a7c15ull; // never 0 (0 reads as "no hash")
|
||||
+ return h;
|
||||
+}
|
||||
+
|
||||
+std::vector<uint64_t> PagedKVManager::compute_block_hashes(const std::vector<int>& token_ids) const {
|
||||
+ std::vector<uint64_t> hashes;
|
||||
+ uint64_t parent = 0; // NONE_HASH analogue
|
||||
+ size_t n_full = token_ids.size() / block_size_;
|
||||
+ for (size_t i = 0; i < n_full; ++i) {
|
||||
+ std::vector<int> blk(token_ids.begin() + i * block_size_,
|
||||
+ token_ids.begin() + (i + 1) * block_size_);
|
||||
+ parent = hash_block(parent, blk);
|
||||
+ hashes.push_back(parent);
|
||||
+ }
|
||||
+ return hashes;
|
||||
+}
|
||||
+
|
||||
+size_t PagedKVManager::get_computed_blocks(const std::vector<uint64_t>& block_hashes) {
|
||||
+ std::vector<KVCacheBlock*> hits;
|
||||
+ for (uint64_t bh : block_hashes) { // stop at first miss (prefix property)
|
||||
+ KVCacheBlock* cb = pool_.get_cached_block(bh);
|
||||
+ if (!cb) break;
|
||||
+ hits.push_back(cb);
|
||||
+ }
|
||||
+ pool_.touch(hits); // ++ref_cnt, pull from free list
|
||||
+ return hits.size() * (size_t)block_size_;
|
||||
+}
|
||||
+
|
||||
+void PagedKVManager::cache_blocks(int seq_id, const std::vector<uint64_t>& block_hashes, size_t num_tokens) {
|
||||
+ auto& req = req_to_blocks_[seq_id];
|
||||
+ size_t n_full = num_tokens / block_size_;
|
||||
+ pool_.cache_full_blocks(req, /*num_cached=*/0, n_full, block_hashes);
|
||||
+}
|
||||
+
|
||||
+} // namespace paged
|
||||
diff --git a/src/paged-kv-manager.h b/src/paged-kv-manager.h
|
||||
new file mode 100644
|
||||
index 000000000..740280a7f
|
||||
--- /dev/null
|
||||
+++ b/src/paged-kv-manager.h
|
||||
@@ -0,0 +1,109 @@
|
||||
+#pragma once
|
||||
+// Paged KV cache block manager for llama.cpp (CPU-first prototype).
|
||||
+//
|
||||
+// Host-side block management is a faithful port of vLLM V1:
|
||||
+// vllm/v1/core/kv_cache_utils.py (KVCacheBlock, FreeKVCacheBlockQueue, hash_block_tokens)
|
||||
+// vllm/v1/core/block_pool.py (BlockPool: get_new_blocks/touch/free/evict/cache_full_blocks)
|
||||
+// vllm/v1/core/single_type_kv_cache_manager.py (allocate_new_blocks, find_longest_cache_hit)
|
||||
+//
|
||||
+// Parity is on behavior/algorithm (block chaining, first-miss stop, ref-counting,
|
||||
+// LRU eviction order), not on exact hash bytes. This unit has zero ggml/llama.cpp
|
||||
+// dependency so it can be unit-tested in isolation.
|
||||
+
|
||||
+#include <cstddef>
|
||||
+#include <cstdint>
|
||||
+#include <vector>
|
||||
+#include <unordered_map>
|
||||
+#include <map>
|
||||
+
|
||||
+namespace paged {
|
||||
+
|
||||
+// vLLM KVCacheBlock (kv_cache_utils.py).
|
||||
+struct KVCacheBlock {
|
||||
+ int32_t block_id = 0;
|
||||
+ int ref_cnt = 0;
|
||||
+ bool has_hash = false; // vLLM: _block_hash is set only when full+cached
|
||||
+ uint64_t block_hash = 0;
|
||||
+ bool is_null = false;
|
||||
+ KVCacheBlock* prev_free = nullptr;
|
||||
+ KVCacheBlock* next_free = nullptr;
|
||||
+
|
||||
+ explicit KVCacheBlock(int32_t id = 0) : block_id(id) {}
|
||||
+ void reset_hash() { has_hash = false; block_hash = 0; }
|
||||
+};
|
||||
+
|
||||
+// Intrusive doubly-linked free list with fake head/tail (vLLM FreeKVCacheBlockQueue).
|
||||
+// O(1) middle removal is required so touch() can pull a warm cached block out of the
|
||||
+// free list when a later request hits its prefix.
|
||||
+class FreeBlockQueue {
|
||||
+public:
|
||||
+ size_t num_free_blocks = 0;
|
||||
+
|
||||
+ explicit FreeBlockQueue(const std::vector<KVCacheBlock*>& blocks);
|
||||
+ KVCacheBlock* popleft();
|
||||
+ std::vector<KVCacheBlock*> popleft_n(size_t n);
|
||||
+ void remove(KVCacheBlock* block);
|
||||
+ void append(KVCacheBlock* block);
|
||||
+ void append_n(const std::vector<KVCacheBlock*>& blocks);
|
||||
+ void prepend_n(const std::vector<KVCacheBlock*>& blocks);
|
||||
+ std::vector<KVCacheBlock*> get_all_free_blocks() const;
|
||||
+
|
||||
+private:
|
||||
+ KVCacheBlock fake_head{-1};
|
||||
+ KVCacheBlock fake_tail{-1};
|
||||
+};
|
||||
+
|
||||
+// vLLM BlockPool (block_pool.py).
|
||||
+class BlockPool {
|
||||
+public:
|
||||
+ KVCacheBlock* null_block = nullptr;
|
||||
+
|
||||
+ BlockPool(int32_t num_blocks, bool enable_caching);
|
||||
+ std::vector<KVCacheBlock*> get_new_blocks(size_t n);
|
||||
+ KVCacheBlock* get_cached_block(uint64_t block_hash);
|
||||
+ void touch(const std::vector<KVCacheBlock*>& blocks);
|
||||
+ void free_blocks(const std::vector<KVCacheBlock*>& ordered_blocks);
|
||||
+ void cache_full_blocks(const std::vector<KVCacheBlock*>& req_blocks,
|
||||
+ size_t num_cached_blocks, size_t num_full_blocks,
|
||||
+ const std::vector<uint64_t>& block_hashes);
|
||||
+ size_t get_num_free_blocks() const { return free_queue_.num_free_blocks; }
|
||||
+
|
||||
+private:
|
||||
+ bool maybe_evict_cached_block(KVCacheBlock* block);
|
||||
+
|
||||
+ bool enable_caching_;
|
||||
+ std::vector<KVCacheBlock> blocks_; // owns all block descriptors
|
||||
+ std::vector<KVCacheBlock*> ptrs_;
|
||||
+ FreeBlockQueue free_queue_;
|
||||
+ // vLLM stores hash -> {block_id: block} to allow duplicate-content blocks; the
|
||||
+ // prototype keeps the last writer (single KV-cache group is sufficient for the wins).
|
||||
+ std::unordered_map<uint64_t, KVCacheBlock*> cached_block_hash_to_block_;
|
||||
+};
|
||||
+
|
||||
+// Allocation + prefix-caching surface, ported from SingleTypeKVCacheManager /
|
||||
+// FullAttentionManager. Single KV-cache group; no extra_keys / eagle / spec-decode.
|
||||
+class PagedKVManager {
|
||||
+public:
|
||||
+ PagedKVManager(int32_t num_blocks, int block_size, bool enable_caching);
|
||||
+
|
||||
+ // Grow seq_id to cover total_tokens slots. Returns false on OOM (free queue empty).
|
||||
+ bool allocate(int seq_id, size_t total_tokens);
|
||||
+ std::vector<int32_t> block_table(int seq_id) const;
|
||||
+ int64_t slot(int seq_id, int pos) const;
|
||||
+ std::vector<int64_t> slot_mapping(int seq_id, const std::vector<int>& positions) const;
|
||||
+ void free(int seq_id);
|
||||
+ int block_size() const { return block_size_; }
|
||||
+
|
||||
+ // Prefix caching (win 3).
|
||||
+ static uint64_t hash_block(uint64_t parent_hash, const std::vector<int>& token_ids);
|
||||
+ std::vector<uint64_t> compute_block_hashes(const std::vector<int>& token_ids) const;
|
||||
+ size_t get_computed_blocks(const std::vector<uint64_t>& block_hashes); // returns num cached tokens
|
||||
+ void cache_blocks(int seq_id, const std::vector<uint64_t>& block_hashes, size_t num_tokens);
|
||||
+
|
||||
+protected:
|
||||
+ int block_size_;
|
||||
+ BlockPool pool_;
|
||||
+ std::map<int, std::vector<KVCacheBlock*>> req_to_blocks_;
|
||||
+};
|
||||
+
|
||||
+} // namespace paged
|
||||
--
|
||||
2.43.0
|
||||
|
||||
@@ -1,75 +0,0 @@
|
||||
From 5c9c709e6c6b07e0399b75fd4e46e752d418a9a8 Mon Sep 17 00:00:00 2001
|
||||
From: Ettore Di Giacinto <mudler@localai.io>
|
||||
Date: Fri, 19 Jun 2026 23:04:17 +0000
|
||||
Subject: [PATCH] paged kv block placement (env LLAMA_KV_PAGED)
|
||||
|
||||
Place each sequence's tokens at permuted, non-contiguous fixed-size block
|
||||
positions in find_slot, proving attention is invariant to physical KV placement
|
||||
(token-identical greedy generation). Default off; single-sequence scope; falls
|
||||
back to the normal allocator. The paged-placement substrate for the gather-read.
|
||||
---
|
||||
src/llama-kv-cache.cpp | 41 +++++++++++++++++++++++++++++++++++++++++
|
||||
1 file changed, 41 insertions(+)
|
||||
|
||||
diff --git a/src/llama-kv-cache.cpp b/src/llama-kv-cache.cpp
|
||||
index 2802103bd..999e2ae61 100644
|
||||
--- a/src/llama-kv-cache.cpp
|
||||
+++ b/src/llama-kv-cache.cpp
|
||||
@@ -11,6 +11,8 @@
|
||||
#include <cstring>
|
||||
#include <limits>
|
||||
#include <map>
|
||||
+#include <numeric>
|
||||
+#include <cstdlib>
|
||||
#include <stdexcept>
|
||||
|
||||
static bool ggml_is_power_of_2(int n) {
|
||||
@@ -1020,6 +1022,45 @@ llama_kv_cache::slot_info llama_kv_cache::find_slot(const llama_ubatch & ubatch,
|
||||
return { };
|
||||
}
|
||||
|
||||
+ // [paged, experimental] Place this sequence's tokens at permuted,
|
||||
+ // non-contiguous fixed-size BLOCK positions instead of a contiguous run.
|
||||
+ // This validates that attention is invariant to physical KV placement -
|
||||
+ // the correctness premise of paged attention. Enabled via LLAMA_KV_PAGED.
|
||||
+ // Single-sequence scope (uses get_used() as the logical base); falls back
|
||||
+ // to the normal allocator if the permuted cells aren't available.
|
||||
+ static const bool paged_mode = (std::getenv("LLAMA_KV_PAGED") != nullptr);
|
||||
+ if (paged_mode) {
|
||||
+ const uint32_t bs = 16; // block size (tokens/block)
|
||||
+ const uint32_t nblk = cells.size() / bs; // blocks in this stream's pool
|
||||
+ if (nblk >= 2) {
|
||||
+ // stride coprime to nblk => block-index permutation is a bijection
|
||||
+ uint32_t k = 1;
|
||||
+ for (uint32_t cand = (nblk / 2) | 1u; cand < nblk; cand += 2) {
|
||||
+ if (std::gcd(cand, nblk) == 1u) { k = cand; break; }
|
||||
+ }
|
||||
+ const uint32_t base = cells.get_used();
|
||||
+ bool ok = true;
|
||||
+ for (uint32_t i = 0; i < n_tokens; ++i) {
|
||||
+ const uint32_t L = base + i;
|
||||
+ const uint32_t b = L / bs;
|
||||
+ const uint32_t off = L % bs;
|
||||
+ if (b >= nblk) { ok = false; break; }
|
||||
+ const uint32_t phys = ((b * k) % nblk) * bs + off; // permuted block
|
||||
+ if (phys >= cells.size() || !cells.is_empty(phys)) { ok = false; break; }
|
||||
+ res.idxs[s].push_back(phys);
|
||||
+ }
|
||||
+ if (ok && res.idxs[s].size() == n_tokens) {
|
||||
+ if (std::getenv("LLAMA_KV_PAGED_DEBUG")) {
|
||||
+ fprintf(stderr, "[paged] seq placed %u tok at cells:", n_tokens);
|
||||
+ for (uint32_t z = 0; z < res.idxs[s].size() && z < 24; ++z) fprintf(stderr, " %u", res.idxs[s][z]);
|
||||
+ fprintf(stderr, " (k=%u nblk=%u base=%u)\n", k, nblk, base);
|
||||
+ }
|
||||
+ continue; // paged placement succeeded for this sequence
|
||||
+ }
|
||||
+ res.idxs[s].clear(); // fall back to the normal allocator
|
||||
+ }
|
||||
+ }
|
||||
+
|
||||
uint32_t n_tested = 0;
|
||||
|
||||
// for continuous slots, we test that all tokens in the ubatch fit, starting from the current head
|
||||
--
|
||||
2.43.0
|
||||
|
||||
@@ -1,370 +0,0 @@
|
||||
From c1de00f4cc1eb0dd25993880bb4c8562be1937d4 Mon Sep 17 00:00:00 2001
|
||||
From: Ettore Di Giacinto <mudler@localai.io>
|
||||
Date: Mon, 22 Jun 2026 10:24:22 +0200
|
||||
Subject: [PATCH] paged gather-read (env LLAMA_KV_PAGED) - patch 0003
|
||||
|
||||
Gather K, V and the kq_mask down to each sequence stream's non-empty cells
|
||||
before build_attn_mha. Position-sorted per stream so the flash-attn online
|
||||
softmax reduction order matches stock byte-for-byte. Multi-stream: one index
|
||||
column per stream over k->ne[3], padded to the max non-empty count with a
|
||||
masked (empty) cell. Gated behind LLAMA_KV_PAGED; no-op when unset.
|
||||
---
|
||||
src/CMakeLists.txt | 1 +
|
||||
src/llama-graph.cpp | 9 ++-
|
||||
src/llama-kv-cache.cpp | 74 ++++++++++++++++++++++++
|
||||
src/llama-kv-cache.h | 11 ++++
|
||||
src/paged-attn.cpp | 128 +++++++++++++++++++++++++++++++++++++++++
|
||||
src/paged-attn.h | 40 +++++++++++++
|
||||
6 files changed, 262 insertions(+), 1 deletion(-)
|
||||
create mode 100644 src/paged-attn.cpp
|
||||
create mode 100644 src/paged-attn.h
|
||||
|
||||
diff --git a/src/CMakeLists.txt b/src/CMakeLists.txt
|
||||
index a030940..58083b3 100644
|
||||
--- a/src/CMakeLists.txt
|
||||
+++ b/src/CMakeLists.txt
|
||||
@@ -25,6 +25,7 @@ add_library(llama
|
||||
llama-kv-cache.cpp
|
||||
llama-kv-cache-iswa.cpp
|
||||
paged-kv-manager.cpp
|
||||
+ paged-attn.cpp
|
||||
llama-kv-cache-dsa.cpp
|
||||
llama-memory.cpp
|
||||
llama-memory-hybrid.cpp
|
||||
diff --git a/src/llama-graph.cpp b/src/llama-graph.cpp
|
||||
index 68c9e60..b59d2a5 100644
|
||||
--- a/src/llama-graph.cpp
|
||||
+++ b/src/llama-graph.cpp
|
||||
@@ -6,6 +6,8 @@
|
||||
#include "llama-cparams.h"
|
||||
|
||||
#include "llama-kv-cache.h"
|
||||
+
|
||||
+#include "paged-attn.h"
|
||||
#include "llama-kv-cache-iswa.h"
|
||||
#include "llama-kv-cache-dsa.h"
|
||||
#include "llama-memory-hybrid.h"
|
||||
@@ -2356,7 +2358,12 @@ ggml_tensor * llm_graph_context::build_attn(
|
||||
ggml_tensor * k = mctx_cur->get_k(ctx0, il);
|
||||
ggml_tensor * v = mctx_cur->get_v(ctx0, il);
|
||||
|
||||
- ggml_tensor * cur = build_attn_mha(q, k, v, kq_b, kq_mask, sinks, v_mla, kq_scale, il);
|
||||
+ // [paged 0003] gather K, V and the mask to the sequence's used cells only
|
||||
+ // (no-op unless env LLAMA_KV_PAGED is set).
|
||||
+ ggml_tensor * kq_mask_g = kq_mask;
|
||||
+ paged_attn::gather(ctx0, res, mctx_cur, &k, &v, &kq_mask_g);
|
||||
+
|
||||
+ ggml_tensor * cur = build_attn_mha(q, k, v, kq_b, kq_mask_g, sinks, v_mla, kq_scale, il);
|
||||
cb(cur, "kqv_out", il);
|
||||
|
||||
if (inp->self_v_rot) {
|
||||
diff --git a/src/llama-kv-cache.cpp b/src/llama-kv-cache.cpp
|
||||
index 999e2ae..30d02d7 100644
|
||||
--- a/src/llama-kv-cache.cpp
|
||||
+++ b/src/llama-kv-cache.cpp
|
||||
@@ -1,4 +1,6 @@
|
||||
#include "llama-kv-cache.h"
|
||||
+#include <vector>
|
||||
+#include <utility>
|
||||
|
||||
#include "llama-impl.h"
|
||||
#include "llama-io.h"
|
||||
@@ -1329,6 +1331,70 @@ ggml_tensor * llama_kv_cache::get_v(ggml_context * ctx, int32_t il, uint32_t n_k
|
||||
ggml_row_size(v->type, kv_size*n_embd_v_gqa)*sinfo.s0);
|
||||
}
|
||||
|
||||
+// [paged 0003] gather-read: enumerate the non-empty cells in [0, n_kv) for the
|
||||
+// single stream addressed by sinfo. With paged placement (patch 0002) these are
|
||||
+// the sequence's scattered block cells; gathering K/V/mask by this index list
|
||||
+// compacts the attention read while preserving every unmasked (token,cell) pair.
|
||||
+uint32_t llama_kv_cache::get_n_gather(uint32_t n_kv, const slot_info & sinfo) const {
|
||||
+ // Multi-stream: the gathered K/V/mask tensors are rectangular [.., n_gather,
|
||||
+ // n_stream], so n_gather is the MAX non-empty count across the batch streams.
|
||||
+ // Streams with fewer cells are padded (see get_gather_idxs) with a masked
|
||||
+ // (empty) cell index, which contributes exp(-inf)=0 and is thus a no-op.
|
||||
+ // K is laid out over physical streams [s0, s1]; index v_cells the same way.
|
||||
+ const uint32_t ns = sinfo.s1 - sinfo.s0 + 1;
|
||||
+ uint32_t mx = 0;
|
||||
+ for (uint32_t j = 0; j < ns; ++j) {
|
||||
+ const auto & cells = v_cells[sinfo.s0 + j];
|
||||
+ const uint32_t n = std::min<uint32_t>(n_kv, cells.size());
|
||||
+ uint32_t cnt = 0;
|
||||
+ for (uint32_t i = 0; i < n; ++i) {
|
||||
+ if (!cells.is_empty(i)) {
|
||||
+ ++cnt;
|
||||
+ }
|
||||
+ }
|
||||
+ mx = std::max(mx, cnt);
|
||||
+ }
|
||||
+ return mx;
|
||||
+}
|
||||
+
|
||||
+void llama_kv_cache::get_gather_idxs(int32_t * dst, uint32_t n_kv, const slot_info & sinfo) const {
|
||||
+ const uint32_t ns = sinfo.s1 - sinfo.s0 + 1;
|
||||
+ const uint32_t n_gather = get_n_gather(n_kv, sinfo);
|
||||
+ // dst is [n_gather, n_stream] (ne0 = n_gather): column s at dst[s*n_gather..].
|
||||
+ for (uint32_t j = 0; j < ns; ++j) {
|
||||
+ const auto & cells = v_cells[sinfo.s0 + j];
|
||||
+ const uint32_t n = std::min<uint32_t>(n_kv, cells.size());
|
||||
+ // Collect the non-empty cells, then order them by token POSITION (not by
|
||||
+ // physical cell index). The attention reduction (flash-attn online
|
||||
+ // softmax, and the non-flash soft_max) runs over cells in array order and
|
||||
+ // is order-sensitive in floating point. Stock (contiguous) placement
|
||||
+ // happens to store cells in position order, so emitting the gathered
|
||||
+ // indices in position order reproduces stock's exact reduction order -
|
||||
+ // making the paged read bit-identical, not merely math-equivalent.
|
||||
+ std::vector<std::pair<llama_pos, int32_t>> pc;
|
||||
+ pc.reserve(n);
|
||||
+ int32_t pad = -1;
|
||||
+ for (uint32_t i = 0; i < n; ++i) {
|
||||
+ if (!cells.is_empty(i)) {
|
||||
+ pc.emplace_back(cells.pos_get(i), (int32_t) i);
|
||||
+ } else if (pad < 0) {
|
||||
+ pad = (int32_t) i; // first empty cell: its mask is -inf -> safe pad
|
||||
+ }
|
||||
+ }
|
||||
+ std::sort(pc.begin(), pc.end());
|
||||
+ int32_t * col = dst + (size_t) j * n_gather;
|
||||
+ for (size_t k = 0; k < pc.size(); ++k) {
|
||||
+ col[k] = pc[k].second;
|
||||
+ }
|
||||
+ // Pad the tail to n_gather with a masked (empty) cell so the rectangular
|
||||
+ // gather drops to zero contribution for streams shorter than the max.
|
||||
+ const int32_t padv = (pad >= 0) ? pad : (pc.empty() ? 0 : pc.back().second);
|
||||
+ for (uint32_t k = (uint32_t) pc.size(); k < n_gather; ++k) {
|
||||
+ col[k] = padv;
|
||||
+ }
|
||||
+ }
|
||||
+}
|
||||
+
|
||||
ggml_tensor * llama_kv_cache::cpy_k(ggml_context * ctx, ggml_tensor * k_cur, ggml_tensor * k_idxs, int32_t il, const slot_info & sinfo) const {
|
||||
GGML_UNUSED(sinfo);
|
||||
|
||||
@@ -2620,6 +2686,14 @@ ggml_tensor * llama_kv_cache_context::get_v(ggml_context * ctx, int32_t il) cons
|
||||
return kv->get_v(ctx, il, n_kv, sinfos[i_cur]);
|
||||
}
|
||||
|
||||
+uint32_t llama_kv_cache_context::get_n_gather() const {
|
||||
+ return kv->get_n_gather(n_kv, sinfos[i_cur]);
|
||||
+}
|
||||
+
|
||||
+void llama_kv_cache_context::get_gather_idxs(int32_t * dst) const {
|
||||
+ kv->get_gather_idxs(dst, n_kv, sinfos[i_cur]);
|
||||
+}
|
||||
+
|
||||
ggml_tensor * llama_kv_cache_context::cpy_k(ggml_context * ctx, ggml_tensor * k_cur, ggml_tensor * k_idxs, int32_t il) const {
|
||||
return kv->cpy_k(ctx, k_cur, k_idxs, il, sinfos[i_cur]);
|
||||
}
|
||||
diff --git a/src/llama-kv-cache.h b/src/llama-kv-cache.h
|
||||
index 3d68f98..494c0fb 100644
|
||||
--- a/src/llama-kv-cache.h
|
||||
+++ b/src/llama-kv-cache.h
|
||||
@@ -171,6 +171,12 @@ public:
|
||||
ggml_tensor * get_k(ggml_context * ctx, int32_t il, uint32_t n_kv, const slot_info & sinfo) const;
|
||||
ggml_tensor * get_v(ggml_context * ctx, int32_t il, uint32_t n_kv, const slot_info & sinfo) const;
|
||||
|
||||
+ // [paged 0003] count / list the non-empty cells in [0, n_kv) per stream of
|
||||
+ // sinfo (position-sorted, padded across streams). Used by paged-attn
|
||||
+ // gather-read. get_n_gather returns the max count across streams.
|
||||
+ uint32_t get_n_gather(uint32_t n_kv, const slot_info & sinfo) const;
|
||||
+ void get_gather_idxs(int32_t * dst, uint32_t n_kv, const slot_info & sinfo) const;
|
||||
+
|
||||
// store k_cur and v_cur in the cache based on the provided head location
|
||||
ggml_tensor * cpy_k(ggml_context * ctx, ggml_tensor * k_cur, ggml_tensor * k_idxs, int32_t il, const slot_info & sinfo) const;
|
||||
ggml_tensor * cpy_v(ggml_context * ctx, ggml_tensor * v_cur, ggml_tensor * v_idxs, int32_t il, const slot_info & sinfo) const;
|
||||
@@ -368,6 +374,11 @@ public:
|
||||
ggml_tensor * get_k(ggml_context * ctx, int32_t il) const;
|
||||
ggml_tensor * get_v(ggml_context * ctx, int32_t il) const;
|
||||
|
||||
+ // [paged 0003] gather-read helpers (delegate to the kv cache for the
|
||||
+ // current ubatch's stream).
|
||||
+ uint32_t get_n_gather() const;
|
||||
+ void get_gather_idxs(int32_t * dst) const;
|
||||
+
|
||||
// store k_cur and v_cur in the cache based on the provided head location
|
||||
// note: the heads in k_cur and v_cur should be laid out contiguously in memory
|
||||
// - k_cur [n_embd_head_k, n_head_k, n_tokens]
|
||||
diff --git a/src/paged-attn.cpp b/src/paged-attn.cpp
|
||||
new file mode 100644
|
||||
index 0000000..ade75e8
|
||||
--- /dev/null
|
||||
+++ b/src/paged-attn.cpp
|
||||
@@ -0,0 +1,128 @@
|
||||
+#include "paged-attn.h"
|
||||
+
|
||||
+#include "llama-graph.h"
|
||||
+#include "llama-kv-cache.h"
|
||||
+
|
||||
+#include "ggml.h"
|
||||
+#include "ggml-backend.h"
|
||||
+
|
||||
+#include <cstdlib>
|
||||
+#include <cstdio>
|
||||
+
|
||||
+namespace paged_attn {
|
||||
+
|
||||
+bool active() {
|
||||
+ static const bool a = (std::getenv("LLAMA_KV_PAGED") != nullptr);
|
||||
+ return a;
|
||||
+}
|
||||
+
|
||||
+static bool debug() {
|
||||
+ static const bool d = (std::getenv("LLAMA_KV_PAGED_DEBUG") != nullptr);
|
||||
+ return d;
|
||||
+}
|
||||
+
|
||||
+namespace {
|
||||
+
|
||||
+// Graph input that, at set_input time, fills an I32 [n_gather, n_stream] tensor
|
||||
+// with each stream's non-empty cell indices (position-sorted, padded with a
|
||||
+// masked/empty cell) by delegating to the kv-cache context. Private to this
|
||||
+// unit; default can_reuse()==false keeps the graph from being reused across
|
||||
+// decodes (n_gather grows every step).
|
||||
+class input_gather_idxs : public llm_graph_input_i {
|
||||
+public:
|
||||
+ input_gather_idxs(const llama_kv_cache_context * mctx, ggml_tensor * idxs)
|
||||
+ : mctx(mctx), idxs(idxs) {}
|
||||
+
|
||||
+ void set_input(const llama_ubatch * ubatch) override {
|
||||
+ GGML_UNUSED(ubatch);
|
||||
+ GGML_ASSERT(idxs && ggml_backend_buffer_is_host(idxs->buffer));
|
||||
+ mctx->get_gather_idxs((int32_t *) idxs->data);
|
||||
+ }
|
||||
+
|
||||
+ const llama_kv_cache_context * mctx;
|
||||
+ ggml_tensor * idxs;
|
||||
+};
|
||||
+
|
||||
+} // namespace
|
||||
+
|
||||
+void gather(ggml_context * ctx0,
|
||||
+ llm_graph_result * res,
|
||||
+ const llama_kv_cache_context * mctx,
|
||||
+ ggml_tensor ** k,
|
||||
+ ggml_tensor ** v,
|
||||
+ ggml_tensor ** kq_mask) {
|
||||
+ if (!active()) {
|
||||
+ return;
|
||||
+ }
|
||||
+
|
||||
+ ggml_tensor * K = *k;
|
||||
+ ggml_tensor * V = *v;
|
||||
+ ggml_tensor * M = *kq_mask;
|
||||
+
|
||||
+ // Number of streams (sequences) in the unified batch. K is laid out
|
||||
+ // [d, h, n_kv, n_stream] and the mask is [n_kv, n_tps, 1, n_stream]; the
|
||||
+ // gather is per-stream (one index column per stream), so a single
|
||||
+ // ggml_get_rows over the stream axis handles 1..N streams uniformly.
|
||||
+ const int64_t n_stream = K->ne[3];
|
||||
+ GGML_ASSERT(M->ne[3] == n_stream);
|
||||
+
|
||||
+ const int64_t n_gather = (int64_t) mctx->get_n_gather();
|
||||
+ if (n_gather <= 0) {
|
||||
+ // Worst-case graph reserve (empty cache) or nothing placed yet: leave
|
||||
+ // the full [0, n_kv) read untouched so buffer sizing stays worst-case.
|
||||
+ return;
|
||||
+ }
|
||||
+
|
||||
+ if (debug()) {
|
||||
+ static int64_t once = 0;
|
||||
+ if (once++ < 2) {
|
||||
+ fprintf(stderr, "[paged-attn] gather n_stream=%lld n_kv=%lld n_gather=%lld\n",
|
||||
+ (long long) n_stream, (long long) K->ne[2], (long long) n_gather);
|
||||
+ }
|
||||
+ }
|
||||
+
|
||||
+ // Per-stream index tensor [n_gather, n_stream], filled at set_input from
|
||||
+ // each stream's non-empty cells. ggml_get_rows broadcasts along ne[1]==
|
||||
+ // n_stream, so column s gathers from stream s of the source.
|
||||
+ ggml_tensor * idx = ggml_new_tensor_2d(ctx0, GGML_TYPE_I32, n_gather, n_stream);
|
||||
+ ggml_set_input(idx);
|
||||
+ res->add_input(llm_graph_input_ptr(new input_gather_idxs(mctx, idx)));
|
||||
+
|
||||
+ // --- gather K: collapse (head_dim, n_head) so cells become the row axis ---
|
||||
+ {
|
||||
+ ggml_tensor * t = ggml_cont(ctx0, K); // [d, h, n_kv, ns]
|
||||
+ t = ggml_reshape_3d(ctx0, t, K->ne[0]*K->ne[1], K->ne[2], n_stream); // [d*h, n_kv, ns]
|
||||
+ t = ggml_get_rows(ctx0, t, idx); // [d*h, n_gather, ns]
|
||||
+ *k = ggml_reshape_4d(ctx0, t, K->ne[0], K->ne[1], n_gather, n_stream); // [d, h, n_gather, ns]
|
||||
+ }
|
||||
+
|
||||
+ // --- gather V ---
|
||||
+ // Normalize to a non-transposed [d, h, n_kv, ns] view first, so the gathered
|
||||
+ // result is contiguous and build_attn_mha sees a consistent v_trans==false.
|
||||
+ {
|
||||
+ const bool v_trans = V->nb[1] > V->nb[2];
|
||||
+ ggml_tensor * vsrc = v_trans
|
||||
+ ? ggml_permute(ctx0, V, 2, 1, 0, 3) // [n_kv, h, d, ns] -> [d, h, n_kv, ns]
|
||||
+ : V; // already [d, h, n_kv, ns]
|
||||
+ ggml_tensor * t = ggml_cont(ctx0, vsrc); // [d, h, n_kv, ns]
|
||||
+ t = ggml_reshape_3d(ctx0, t, vsrc->ne[0]*vsrc->ne[1], vsrc->ne[2], n_stream); // [d*h, n_kv, ns]
|
||||
+ t = ggml_get_rows(ctx0, t, idx); // [d*h, n_gather, ns]
|
||||
+ *v = ggml_reshape_4d(ctx0, t, vsrc->ne[0], vsrc->ne[1], n_gather, n_stream); // [d, h, n_gather, ns]
|
||||
+ }
|
||||
+
|
||||
+ // --- gather mask (cells are ne0): transpose so cells become the row axis,
|
||||
+ // gather per stream, transpose back ---
|
||||
+ {
|
||||
+ ggml_tensor * m = ggml_reshape_3d(ctx0, M, M->ne[0], M->ne[1], n_stream); // [n_kv, n_tps, ns]
|
||||
+ m = ggml_cont(ctx0, ggml_transpose(ctx0, m)); // [n_tps, n_kv, ns]
|
||||
+ m = ggml_get_rows(ctx0, m, idx); // [n_tps, n_gather, ns] (F32)
|
||||
+ m = ggml_cont(ctx0, ggml_transpose(ctx0, m)); // [n_gather, n_tps, ns]
|
||||
+ m = ggml_reshape_4d(ctx0, m, n_gather, M->ne[1], 1, n_stream);
|
||||
+ if (M->type != m->type) {
|
||||
+ m = ggml_cast(ctx0, m, M->type); // flash-attn requires an F16 mask
|
||||
+ }
|
||||
+ *kq_mask = m;
|
||||
+ }
|
||||
+}
|
||||
+
|
||||
+} // namespace paged_attn
|
||||
diff --git a/src/paged-attn.h b/src/paged-attn.h
|
||||
new file mode 100644
|
||||
index 0000000..c5b7bd7
|
||||
--- /dev/null
|
||||
+++ b/src/paged-attn.h
|
||||
@@ -0,0 +1,41 @@
|
||||
+#pragma once
|
||||
+// Paged attention gather-read (patch 0003, experimental).
|
||||
+//
|
||||
+// Companion to the paged block placement in llama_kv_cache::find_slot (patch
|
||||
+// 0002). Patch 0002 places a sequence's tokens at permuted, non-contiguous
|
||||
+// fixed-size block cells, but attention still reads the whole [0, n_kv) window
|
||||
+// (empty cells masked to -inf). This unit compacts that read: it gathers K, V
|
||||
+// and the kq_mask down to ONLY the sequence's used (non-empty) cells before
|
||||
+// build_attn_mha.
|
||||
+//
|
||||
+// Correctness: attention is permutation-invariant over the KV set, and dropping
|
||||
+// already-masked empty cells removes only exp(-inf)=0 terms - so greedy output
|
||||
+// is identical to stock. Gated behind env LLAMA_KV_PAGED; a no-op when unset.
|
||||
+//
|
||||
+// All logic lives here to keep the core files additive: build_attn gets one
|
||||
+// call, llama_kv_cache_context gets two thin accessors, CMake gets one line.
|
||||
+
|
||||
+#include <cstddef>
|
||||
+#include <cstdint>
|
||||
+
|
||||
+struct ggml_context;
|
||||
+struct ggml_tensor;
|
||||
+class llm_graph_result;
|
||||
+class llama_kv_cache_context;
|
||||
+
|
||||
+namespace paged_attn {
|
||||
+
|
||||
+// true iff env LLAMA_KV_PAGED is set (evaluated once).
|
||||
+bool active();
|
||||
+
|
||||
+// Gather K, V and the kq_mask down to the current sequence's non-empty cells.
|
||||
+// No-op (returns immediately) unless active(). On return *k, *v and *kq_mask
|
||||
+// point at the compacted tensors; pass them straight to build_attn_mha.
|
||||
+void gather(ggml_context * ctx0,
|
||||
+ llm_graph_result * res,
|
||||
+ const llama_kv_cache_context * mctx,
|
||||
+ ggml_tensor ** k,
|
||||
+ ggml_tensor ** v,
|
||||
+ ggml_tensor ** kq_mask);
|
||||
+
|
||||
+} // namespace paged_attn
|
||||
--
|
||||
2.43.0
|
||||
|
||||
@@ -1,298 +0,0 @@
|
||||
From 7c294973de28d1ac991505638d726acfb371d541 Mon Sep 17 00:00:00 2001
|
||||
From: Ettore Di Giacinto <mudler@localai.io>
|
||||
Date: Mon, 22 Jun 2026 10:50:35 +0200
|
||||
Subject: [PATCH] paged on-demand block allocation (env LLAMA_KV_PAGED) - patch
|
||||
0004
|
||||
|
||||
Drive the paged placement in find_slot through the vendored PagedKVManager
|
||||
(patch 0001) instead of a fixed full-pool permutation. Blocks are popped from a
|
||||
free pool on demand as the sequence crosses block boundaries (peak << full
|
||||
reservation) and returned on sequence end (seq_rm full removal / clear). One
|
||||
manager per (kv-cache, stream); all state lives in the new src/paged-alloc unit,
|
||||
so the core kv-cache struct is untouched - find_slot/clear/seq_rm gain only a
|
||||
gated call. Default off; stock path byte-identical.
|
||||
---
|
||||
src/CMakeLists.txt | 1 +
|
||||
src/llama-kv-cache.cpp | 69 +++++++++++++++++----------
|
||||
src/paged-alloc.cpp | 106 +++++++++++++++++++++++++++++++++++++++++
|
||||
src/paged-alloc.h | 39 +++++++++++++++
|
||||
4 files changed, 190 insertions(+), 25 deletions(-)
|
||||
create mode 100644 src/paged-alloc.cpp
|
||||
create mode 100644 src/paged-alloc.h
|
||||
|
||||
diff --git a/src/CMakeLists.txt b/src/CMakeLists.txt
|
||||
index 58083b3..4d9d7d1 100644
|
||||
--- a/src/CMakeLists.txt
|
||||
+++ b/src/CMakeLists.txt
|
||||
@@ -26,6 +26,7 @@ add_library(llama
|
||||
llama-kv-cache-iswa.cpp
|
||||
paged-kv-manager.cpp
|
||||
paged-attn.cpp
|
||||
+ paged-alloc.cpp
|
||||
llama-kv-cache-dsa.cpp
|
||||
llama-memory.cpp
|
||||
llama-memory-hybrid.cpp
|
||||
diff --git a/src/llama-kv-cache.cpp b/src/llama-kv-cache.cpp
|
||||
index 30d02d7..1125d9a 100644
|
||||
--- a/src/llama-kv-cache.cpp
|
||||
+++ b/src/llama-kv-cache.cpp
|
||||
@@ -1,4 +1,5 @@
|
||||
#include "llama-kv-cache.h"
|
||||
+#include "paged-alloc.h"
|
||||
#include <vector>
|
||||
#include <utility>
|
||||
|
||||
@@ -381,6 +382,11 @@ llama_kv_cache::llama_kv_cache(
|
||||
}
|
||||
|
||||
void llama_kv_cache::clear(bool data) {
|
||||
+ // [paged 0004] return all on-demand blocks to the pool on cache clear.
|
||||
+ if (paged_alloc::active()) {
|
||||
+ paged_alloc::release_all(this);
|
||||
+ }
|
||||
+
|
||||
for (uint32_t s = 0; s < n_stream; ++s) {
|
||||
v_cells[s].reset();
|
||||
v_heads[s] = 0;
|
||||
@@ -409,6 +415,16 @@ bool llama_kv_cache::seq_rm(llama_seq_id seq_id, llama_pos p0, llama_pos p1) {
|
||||
p1 = std::numeric_limits<llama_pos>::max();
|
||||
}
|
||||
|
||||
+ // [paged 0004] free a stream's on-demand blocks when its whole sequence is
|
||||
+ // removed (sequence end), so they return to the pool for reuse.
|
||||
+ if (paged_alloc::active() && p0 == 0 && p1 == std::numeric_limits<llama_pos>::max()) {
|
||||
+ if (seq_id >= 0) {
|
||||
+ paged_alloc::release(this, (int) seq_to_stream[seq_id]);
|
||||
+ } else {
|
||||
+ paged_alloc::release_all(this);
|
||||
+ }
|
||||
+ }
|
||||
+
|
||||
if (seq_id >= 0) {
|
||||
auto & cells = v_cells[seq_to_stream[seq_id]];
|
||||
auto & head = v_heads[seq_to_stream[seq_id]];
|
||||
@@ -1030,36 +1046,39 @@ llama_kv_cache::slot_info llama_kv_cache::find_slot(const llama_ubatch & ubatch,
|
||||
// the correctness premise of paged attention. Enabled via LLAMA_KV_PAGED.
|
||||
// Single-sequence scope (uses get_used() as the logical base); falls back
|
||||
// to the normal allocator if the permuted cells aren't available.
|
||||
- static const bool paged_mode = (std::getenv("LLAMA_KV_PAGED") != nullptr);
|
||||
- if (paged_mode) {
|
||||
+ // [paged 0004] On-demand block allocation. Patch 0002 proved attention is
|
||||
+ // invariant to physical KV placement; here that placement is driven by
|
||||
+ // the vendored PagedKVManager (patch 0001): blocks are popped from a free
|
||||
+ // pool only as the sequence crosses block boundaries (peak << full
|
||||
+ // reservation) and returned on sequence end. Enabled via LLAMA_KV_PAGED;
|
||||
+ // falls back to the normal allocator on pool exhaustion or any conflict.
|
||||
+ if (paged_alloc::active()) {
|
||||
const uint32_t bs = 16; // block size (tokens/block)
|
||||
- const uint32_t nblk = cells.size() / bs; // blocks in this stream's pool
|
||||
+ const uint32_t nblk = cells.size() / bs; // this stream's block budget
|
||||
if (nblk >= 2) {
|
||||
- // stride coprime to nblk => block-index permutation is a bijection
|
||||
- uint32_t k = 1;
|
||||
- for (uint32_t cand = (nblk / 2) | 1u; cand < nblk; cand += 2) {
|
||||
- if (std::gcd(cand, nblk) == 1u) { k = cand; break; }
|
||||
- }
|
||||
const uint32_t base = cells.get_used();
|
||||
- bool ok = true;
|
||||
- for (uint32_t i = 0; i < n_tokens; ++i) {
|
||||
- const uint32_t L = base + i;
|
||||
- const uint32_t b = L / bs;
|
||||
- const uint32_t off = L % bs;
|
||||
- if (b >= nblk) { ok = false; break; }
|
||||
- const uint32_t phys = ((b * k) % nblk) * bs + off; // permuted block
|
||||
- if (phys >= cells.size() || !cells.is_empty(phys)) { ok = false; break; }
|
||||
- res.idxs[s].push_back(phys);
|
||||
- }
|
||||
- if (ok && res.idxs[s].size() == n_tokens) {
|
||||
- if (std::getenv("LLAMA_KV_PAGED_DEBUG")) {
|
||||
- fprintf(stderr, "[paged] seq placed %u tok at cells:", n_tokens);
|
||||
- for (uint32_t z = 0; z < res.idxs[s].size() && z < 24; ++z) fprintf(stderr, " %u", res.idxs[s][z]);
|
||||
- fprintf(stderr, " (k=%u nblk=%u base=%u)\n", k, nblk, base);
|
||||
+ const int strm = (int) seq_to_stream[seq_id];
|
||||
+ std::vector<uint32_t> placed;
|
||||
+ if (paged_alloc::place(this, strm, base, n_tokens, bs, nblk, placed)) {
|
||||
+ bool ok = (placed.size() == n_tokens);
|
||||
+ for (uint32_t i = 0; ok && i < n_tokens; ++i) {
|
||||
+ if (placed[i] >= cells.size() || !cells.is_empty(placed[i])) {
|
||||
+ ok = false;
|
||||
+ }
|
||||
+ }
|
||||
+ if (ok) {
|
||||
+ for (uint32_t phys : placed) {
|
||||
+ res.idxs[s].push_back(phys);
|
||||
+ }
|
||||
+ if (std::getenv("LLAMA_KV_PAGED_DEBUG")) {
|
||||
+ fprintf(stderr, "[paged] stream %d placed %u tok at cells:", strm, n_tokens);
|
||||
+ for (uint32_t z = 0; z < res.idxs[s].size() && z < 24; ++z) fprintf(stderr, " %u", res.idxs[s][z]);
|
||||
+ fprintf(stderr, " (nblk=%u base=%u)\n", nblk, base);
|
||||
+ }
|
||||
+ continue; // on-demand paged placement succeeded
|
||||
}
|
||||
- continue; // paged placement succeeded for this sequence
|
||||
+ res.idxs[s].clear(); // fall back to the normal allocator
|
||||
}
|
||||
- res.idxs[s].clear(); // fall back to the normal allocator
|
||||
}
|
||||
}
|
||||
|
||||
diff --git a/src/paged-alloc.cpp b/src/paged-alloc.cpp
|
||||
new file mode 100644
|
||||
index 0000000..1d13f9c
|
||||
--- /dev/null
|
||||
+++ b/src/paged-alloc.cpp
|
||||
@@ -0,0 +1,106 @@
|
||||
+#include "paged-alloc.h"
|
||||
+#include "paged-kv-manager.h"
|
||||
+
|
||||
+#include <cstdlib>
|
||||
+#include <cstdio>
|
||||
+#include <map>
|
||||
+#include <memory>
|
||||
+#include <utility>
|
||||
+
|
||||
+namespace paged_alloc {
|
||||
+
|
||||
+bool active() {
|
||||
+ static const bool a = (std::getenv("LLAMA_KV_PAGED") != nullptr);
|
||||
+ return a;
|
||||
+}
|
||||
+
|
||||
+static bool debug() {
|
||||
+ static const bool d = (std::getenv("LLAMA_KV_PAGED_DEBUG") != nullptr);
|
||||
+ return d;
|
||||
+}
|
||||
+
|
||||
+namespace {
|
||||
+
|
||||
+using key_t = std::pair<const void *, int>;
|
||||
+
|
||||
+// One PagedKVManager per (kv-cache, stream): each stream owns a separate
|
||||
+// physical pool of cells.size() cells, so a manager's block ids map directly to
|
||||
+// cell ranges within that stream's pool. The internal request id is always 0.
|
||||
+std::map<key_t, std::unique_ptr<paged::PagedKVManager>> g_managers;
|
||||
+
|
||||
+paged::PagedKVManager * get_mgr(const void * cache, int stream,
|
||||
+ uint32_t pool_blocks, uint32_t block_size) {
|
||||
+ const key_t k{cache, stream};
|
||||
+ auto it = g_managers.find(k);
|
||||
+ if (it == g_managers.end()) {
|
||||
+ // enable_caching=false: prefix caching is a later patch; 0004 exercises
|
||||
+ // only on-demand allocate / free.
|
||||
+ auto mgr = std::make_unique<paged::PagedKVManager>(
|
||||
+ (int32_t) pool_blocks, (int) block_size, /*enable_caching=*/false);
|
||||
+ it = g_managers.emplace(k, std::move(mgr)).first;
|
||||
+ }
|
||||
+ return it->second.get();
|
||||
+}
|
||||
+
|
||||
+} // namespace
|
||||
+
|
||||
+bool place(const void * cache, int stream, uint32_t base, uint32_t n_tokens,
|
||||
+ uint32_t block_size, uint32_t pool_blocks,
|
||||
+ std::vector<uint32_t> & out) {
|
||||
+ if (n_tokens == 0) {
|
||||
+ return true;
|
||||
+ }
|
||||
+
|
||||
+ paged::PagedKVManager * mgr = get_mgr(cache, stream, pool_blocks, block_size);
|
||||
+
|
||||
+ const size_t before = mgr->block_table(0).size();
|
||||
+
|
||||
+ // Grow the request to cover the highest logical position. The manager pops
|
||||
+ // free blocks only for the boundaries actually crossed - that is the on-
|
||||
+ // demand behavior; an already-covered range adds nothing.
|
||||
+ if (!mgr->allocate(0, (size_t) base + n_tokens)) {
|
||||
+ return false; // pool exhausted -> caller falls back to the stock path
|
||||
+ }
|
||||
+
|
||||
+ out.reserve(out.size() + n_tokens);
|
||||
+ for (uint32_t i = 0; i < n_tokens; ++i) {
|
||||
+ const int64_t s = mgr->slot(0, (int) (base + i));
|
||||
+ out.push_back((uint32_t) s);
|
||||
+ }
|
||||
+
|
||||
+ if (debug()) {
|
||||
+ const size_t after = mgr->block_table(0).size();
|
||||
+ if (after != before) {
|
||||
+ fprintf(stderr,
|
||||
+ "[paged-alloc] cache=%p stream=%d grew %zu->%zu blocks "
|
||||
+ "(budget=%u; base=%u +%u tok)\n",
|
||||
+ cache, stream, before, after, pool_blocks, base, n_tokens);
|
||||
+ }
|
||||
+ }
|
||||
+
|
||||
+ return true;
|
||||
+}
|
||||
+
|
||||
+void release(const void * cache, int stream) {
|
||||
+ auto it = g_managers.find({cache, stream});
|
||||
+ if (it == g_managers.end()) {
|
||||
+ return;
|
||||
+ }
|
||||
+ it->second->free(0);
|
||||
+ g_managers.erase(it);
|
||||
+ if (debug()) {
|
||||
+ fprintf(stderr, "[paged-alloc] released cache=%p stream=%d\n", cache, stream);
|
||||
+ }
|
||||
+}
|
||||
+
|
||||
+void release_all(const void * cache) {
|
||||
+ for (auto it = g_managers.begin(); it != g_managers.end(); ) {
|
||||
+ if (it->first.first == cache) {
|
||||
+ it = g_managers.erase(it);
|
||||
+ } else {
|
||||
+ ++it;
|
||||
+ }
|
||||
+ }
|
||||
+}
|
||||
+
|
||||
+} // namespace paged_alloc
|
||||
diff --git a/src/paged-alloc.h b/src/paged-alloc.h
|
||||
new file mode 100644
|
||||
index 0000000..bf66665
|
||||
--- /dev/null
|
||||
+++ b/src/paged-alloc.h
|
||||
@@ -0,0 +1,39 @@
|
||||
+#pragma once
|
||||
+// On-demand paged KV block allocation (patch 0004, experimental).
|
||||
+//
|
||||
+// Backs the paged placement in llama_kv_cache::find_slot (patch 0002) with the
|
||||
+// vendored host-side PagedKVManager (patch 0001). Instead of mapping a
|
||||
+// sequence's logical positions onto a fixed full-pool permutation, blocks are
|
||||
+// popped from a free pool ON DEMAND as the sequence crosses block boundaries,
|
||||
+// and returned to the pool on sequence end. This is where the paged memory-
|
||||
+// capacity benefit begins: a short sequence holds only a few blocks, not the
|
||||
+// whole reserved window.
|
||||
+//
|
||||
+// Gated behind env LLAMA_KV_PAGED; a no-op when unset. All state lives in this
|
||||
+// unit (a static registry keyed by kv-cache + stream), so the core kv-cache
|
||||
+// struct stays untouched - find_slot only gains a gated call.
|
||||
+
|
||||
+#include <cstdint>
|
||||
+#include <vector>
|
||||
+
|
||||
+namespace paged_alloc {
|
||||
+
|
||||
+// true iff env LLAMA_KV_PAGED is set (evaluated once).
|
||||
+bool active();
|
||||
+
|
||||
+// Place n_tokens logical positions [base, base+n_tokens) of one stream on
|
||||
+// demand, appending their physical cell indices to `out`. pool_blocks =
|
||||
+// cells.size()/block_size is this stream's block budget. Returns false (leaving
|
||||
+// `out` unchanged) on pool exhaustion, so the caller falls back to the stock
|
||||
+// allocator. The caller still validates each returned cell is empty.
|
||||
+bool place(const void * cache, int stream, uint32_t base, uint32_t n_tokens,
|
||||
+ uint32_t block_size, uint32_t pool_blocks,
|
||||
+ std::vector<uint32_t> & out);
|
||||
+
|
||||
+// Return a stream's blocks to the pool (sequence end).
|
||||
+void release(const void * cache, int stream);
|
||||
+
|
||||
+// Return every stream's blocks for a kv-cache (clear() / teardown).
|
||||
+void release_all(const void * cache);
|
||||
+
|
||||
+} // namespace paged_alloc
|
||||
--
|
||||
2.43.0
|
||||
|
||||
@@ -1,143 +0,0 @@
|
||||
From 141029beec609e87f24f6f6bba3ec842d7037862 Mon Sep 17 00:00:00 2001
|
||||
From: Ettore Di Giacinto <mudler@localai.io>
|
||||
Date: Mon, 22 Jun 2026 12:13:44 +0200
|
||||
Subject: [PATCH] paged cross-request prefix caching (env LLAMA_KV_PAGED) -
|
||||
patch 0006
|
||||
|
||||
Add host-side cross-request prefix sharing to the vendored PagedKVManager
|
||||
(patches 0001-0004): on placement, hash a new sequence prefix blocks, reuse the
|
||||
matching cached physical blocks (ref_cnt++) for the shared prefix and allocate
|
||||
fresh blocks only for the divergent suffix. A shared block is freed only at
|
||||
ref 0; copy-on-write privatises a still-shared (ref>1) block before a divergent
|
||||
write so co-owners stay byte-correct. All logic lives in the vendored
|
||||
src/paged-kv-manager unit (place_with_prefix / cow_block / ref-counting); the
|
||||
core kv-cache files are untouched. Default off; gated behind LLAMA_KV_PAGED.
|
||||
|
||||
Wiring the physical-cell reuse into find_slot so the engine itself skips
|
||||
recompute needs core seq-membership changes and is left to a later patch.
|
||||
|
||||
Assisted-by: Claude:opus-4.8 [Claude Code]
|
||||
Signed-off-by: Ettore Di Giacinto <mudler@localai.io>
|
||||
---
|
||||
src/paged-kv-manager.cpp | 65 ++++++++++++++++++++++++++++++++++++++++
|
||||
src/paged-kv-manager.h | 23 ++++++++++++++
|
||||
2 files changed, 88 insertions(+)
|
||||
|
||||
diff --git a/src/paged-kv-manager.cpp b/src/paged-kv-manager.cpp
|
||||
index ca0dcd8..4c6ee4c 100644
|
||||
--- a/src/paged-kv-manager.cpp
|
||||
+++ b/src/paged-kv-manager.cpp
|
||||
@@ -293,4 +293,69 @@ void PagedKVManager::cache_blocks(int seq_id, const std::vector<uint64_t>& block
|
||||
pool_.cache_full_blocks(req, /*num_cached=*/0, n_full, block_hashes);
|
||||
}
|
||||
|
||||
+// ---------------------------------------------------------------------------
|
||||
+// Cross-request prefix caching + copy-on-write (patch 0006)
|
||||
+// ---------------------------------------------------------------------------
|
||||
+
|
||||
+size_t PagedKVManager::place_with_prefix(int seq_id, const std::vector<int>& token_ids) {
|
||||
+ auto& req = req_to_blocks_[seq_id];
|
||||
+
|
||||
+ // Longest cached prefix: hash the full blocks and stop at the first miss.
|
||||
+ // A block hash transitively encodes its whole prefix (FNV chaining), so the
|
||||
+ // first miss bounds the reusable prefix (vLLM find_longest_cache_hit).
|
||||
+ const std::vector<uint64_t> hashes = compute_block_hashes(token_ids);
|
||||
+ std::vector<KVCacheBlock*> hits;
|
||||
+ for (uint64_t bh : hashes) {
|
||||
+ KVCacheBlock* cb = pool_.get_cached_block(bh);
|
||||
+ if (!cb) break;
|
||||
+ hits.push_back(cb);
|
||||
+ }
|
||||
+
|
||||
+ // Reuse: ++ref_cnt (pulling warm blocks back out of the free list) then
|
||||
+ // splice the shared physical blocks into this sequence's block table.
|
||||
+ pool_.touch(hits);
|
||||
+ req.insert(req.end(), hits.begin(), hits.end());
|
||||
+
|
||||
+ // Allocate fresh blocks only for the divergent suffix.
|
||||
+ const size_t need = cdiv(token_ids.size(), block_size_);
|
||||
+ if (need > req.size()) {
|
||||
+ const size_t add = need - req.size();
|
||||
+ if (add > pool_.get_num_free_blocks()) {
|
||||
+ // OOM: roll the sequence back (un-touch the shared prefix so no ref
|
||||
+ // leaks) and report no placement; the caller falls back to stock.
|
||||
+ std::vector<KVCacheBlock*> ordered(req.rbegin(), req.rend());
|
||||
+ pool_.free_blocks(ordered);
|
||||
+ req.clear();
|
||||
+ return 0;
|
||||
+ }
|
||||
+ auto nb = pool_.get_new_blocks(add);
|
||||
+ req.insert(req.end(), nb.begin(), nb.end());
|
||||
+ }
|
||||
+ return hits.size();
|
||||
+}
|
||||
+
|
||||
+std::pair<int32_t, int32_t> PagedKVManager::cow_block(int seq_id, size_t bi) {
|
||||
+ auto& req = req_to_blocks_.at(seq_id);
|
||||
+ KVCacheBlock* old = req.at(bi);
|
||||
+ if (old->ref_cnt <= 1) {
|
||||
+ return { old->block_id, old->block_id }; // already private - no copy
|
||||
+ }
|
||||
+ // Private copy for this sequence. get_new_blocks sets the fresh block's
|
||||
+ // ref_cnt to 1; free_blocks decrements the shared block, which stays >0 so
|
||||
+ // it is NOT returned to the pool and the other owners are left untouched.
|
||||
+ KVCacheBlock* fresh = pool_.get_new_blocks(1).front();
|
||||
+ pool_.free_blocks({ old });
|
||||
+ req[bi] = fresh;
|
||||
+ return { old->block_id, fresh->block_id };
|
||||
+}
|
||||
+
|
||||
+int PagedKVManager::block_ref_cnt_at(int seq_id, size_t bi) const {
|
||||
+ return req_to_blocks_.at(seq_id).at(bi)->ref_cnt;
|
||||
+}
|
||||
+
|
||||
+size_t PagedKVManager::num_blocks(int seq_id) const {
|
||||
+ auto it = req_to_blocks_.find(seq_id);
|
||||
+ return it == req_to_blocks_.end() ? 0 : it->second.size();
|
||||
+}
|
||||
+
|
||||
} // namespace paged
|
||||
diff --git a/src/paged-kv-manager.h b/src/paged-kv-manager.h
|
||||
index 740280a..34decbc 100644
|
||||
--- a/src/paged-kv-manager.h
|
||||
+++ b/src/paged-kv-manager.h
|
||||
@@ -14,6 +14,7 @@
|
||||
#include <vector>
|
||||
#include <unordered_map>
|
||||
#include <map>
|
||||
+#include <utility>
|
||||
|
||||
namespace paged {
|
||||
|
||||
@@ -99,6 +100,28 @@ public:
|
||||
size_t get_computed_blocks(const std::vector<uint64_t>& block_hashes); // returns num cached tokens
|
||||
void cache_blocks(int seq_id, const std::vector<uint64_t>& block_hashes, size_t num_tokens);
|
||||
|
||||
+ // Cross-request prefix caching + copy-on-write (patch 0006).
|
||||
+ //
|
||||
+ // Splice the longest cached prefix of token_ids into seq_id (reuse the
|
||||
+ // shared physical blocks, ref_cnt++ so a block frees only at ref 0) and
|
||||
+ // allocate fresh blocks only for the divergent suffix. Returns the number of
|
||||
+ // shared (reused) blocks; the caller skips recomputing those tokens. On pool
|
||||
+ // exhaustion the sequence is rolled back (no ref leak) and 0 is returned.
|
||||
+ size_t place_with_prefix(int seq_id, const std::vector<int>& token_ids);
|
||||
+
|
||||
+ // Copy-on-write the block at logical index bi of seq_id. If that block is
|
||||
+ // shared (ref_cnt>1), allocate a fresh private block, drop this seq's ref on
|
||||
+ // the shared one (other owners keep it, content untouched) and install the
|
||||
+ // fresh block at bi. Returns {old_block_id, new_block_id}; new==old when the
|
||||
+ // block was already private (ref_cnt<=1) and no copy is needed. The caller
|
||||
+ // copies the physical cell contents old_block_id -> new_block_id.
|
||||
+ std::pair<int32_t, int32_t> cow_block(int seq_id, size_t bi);
|
||||
+
|
||||
+ // Introspection for the prefix-share gate (debug/tests).
|
||||
+ int block_ref_cnt_at(int seq_id, size_t bi) const;
|
||||
+ size_t num_blocks(int seq_id) const;
|
||||
+ size_t num_free_blocks() const { return pool_.get_num_free_blocks(); }
|
||||
+
|
||||
protected:
|
||||
int block_size_;
|
||||
BlockPool pool_;
|
||||
--
|
||||
2.43.0
|
||||
|
||||
@@ -1,534 +0,0 @@
|
||||
From da20c1c0571e84bc76202d915d4bb82892a3392b Mon Sep 17 00:00:00 2001
|
||||
From: Ettore Di Giacinto <mudler@localai.io>
|
||||
Date: Mon, 22 Jun 2026 12:46:28 +0200
|
||||
Subject: [PATCH] paged engine prefix recompute-skip (env LLAMA_KV_PAGED) -
|
||||
patch 0007
|
||||
|
||||
Wire the host-side cross-request prefix cache (patch 0006) into the engine so a
|
||||
new sequence physically SHARES the cached prefix blocks and skips recomputing the
|
||||
shared prefix - the actual compute win that 0006 (which only proved the host-side
|
||||
machinery + realised reuse via the stock seq_cp) did not yet deliver from the
|
||||
paged path itself.
|
||||
|
||||
Mechanism (all gated behind LLAMA_KV_PAGED; default off, stock byte-identical):
|
||||
|
||||
* paged-alloc reworked from a per-stream, request-0, destroyed-on-free manager
|
||||
into ONE persistent caching PagedKVManager per (kv-cache, stream) whose
|
||||
requests are keyed by the real llama_seq_id. free(seq) now releases exactly
|
||||
one sequence, so ref-counted shared blocks survive while another sharer holds
|
||||
them. New seams: share_prefix (place_with_prefix -> shared prefix tokens),
|
||||
slot, commit (publish a sequence into the content cache), ref-counted release,
|
||||
plus ref/num-free introspection.
|
||||
|
||||
* Two gated llama_kv_cache methods (the core seq-membership handling 0007 needs):
|
||||
paged_prefix_share() reuses the longest cached content prefix for a sequence
|
||||
and marks the shared physical cells as belonging to it (cells.seq_add) so the
|
||||
engine's attention mask includes the already-computed prefix KV; the caller
|
||||
then decodes ONLY the divergent suffix. paged_prefix_commit() publishes a
|
||||
sequence's full blocks for later reuse.
|
||||
|
||||
* find_slot's paged branch anchors placement on each sequence's own logical base
|
||||
(ubatch.pos) and keys the manager request by seq_id, so an independently-freed
|
||||
sequence and a shared prefix coexist in one unified pool. seq_rm/clear free
|
||||
per-sequence (ref-counted) instead of nuking the whole stream.
|
||||
|
||||
* paged-prefix-api: a thin gated shim so a caller holding only the public
|
||||
llama.h can reach the seam and the introspection without the internal headers.
|
||||
|
||||
Core existing-file touch: src/llama-kv-cache.{cpp,h}, +71 -3. Everything else is
|
||||
additive vendored units. Verified on Qwen3-0.6B-Q8_0 (CPU, unified cache): a
|
||||
sequence B sharing A's prefix decodes greedy tokens byte-identical to B from
|
||||
scratch with the prefill computing ONLY the suffix (32 prefix tokens skipped) at
|
||||
a block boundary AND mid-block; the shared block carries ref_cnt 2 while both
|
||||
hold it, drops to 1 when one sharer is removed (survivor intact, re-shareable, no
|
||||
use-after-free) and returns to the pool only when all sharers are freed. The
|
||||
0004 serving gate (unified and non-unified) stays byte-identical stock vs paged.
|
||||
|
||||
Assisted-by: Claude:opus-4.8 [Claude Code]
|
||||
Signed-off-by: Ettore Di Giacinto <mudler@localai.io>
|
||||
---
|
||||
src/CMakeLists.txt | 1 +
|
||||
src/llama-kv-cache.cpp | 66 +++++++++++++++++++++++--
|
||||
src/llama-kv-cache.h | 8 +++
|
||||
src/paged-alloc.cpp | 104 ++++++++++++++++++++++++++++++---------
|
||||
src/paged-alloc.h | 69 +++++++++++++++++++-------
|
||||
src/paged-prefix-api.cpp | 48 ++++++++++++++++++
|
||||
src/paged-prefix-api.h | 27 ++++++++++
|
||||
7 files changed, 280 insertions(+), 43 deletions(-)
|
||||
create mode 100644 src/paged-prefix-api.cpp
|
||||
create mode 100644 src/paged-prefix-api.h
|
||||
|
||||
diff --git a/src/CMakeLists.txt b/src/CMakeLists.txt
|
||||
index 4d9d7d1..432f42d 100644
|
||||
--- a/src/CMakeLists.txt
|
||||
+++ b/src/CMakeLists.txt
|
||||
@@ -27,6 +27,7 @@ add_library(llama
|
||||
paged-kv-manager.cpp
|
||||
paged-attn.cpp
|
||||
paged-alloc.cpp
|
||||
+ paged-prefix-api.cpp
|
||||
llama-kv-cache-dsa.cpp
|
||||
llama-memory.cpp
|
||||
llama-memory-hybrid.cpp
|
||||
diff --git a/src/llama-kv-cache.cpp b/src/llama-kv-cache.cpp
|
||||
index 1125d9a..7510ff9 100644
|
||||
--- a/src/llama-kv-cache.cpp
|
||||
+++ b/src/llama-kv-cache.cpp
|
||||
@@ -419,7 +419,7 @@ bool llama_kv_cache::seq_rm(llama_seq_id seq_id, llama_pos p0, llama_pos p1) {
|
||||
// removed (sequence end), so they return to the pool for reuse.
|
||||
if (paged_alloc::active() && p0 == 0 && p1 == std::numeric_limits<llama_pos>::max()) {
|
||||
if (seq_id >= 0) {
|
||||
- paged_alloc::release(this, (int) seq_to_stream[seq_id]);
|
||||
+ paged_alloc::release(this, (int) seq_to_stream[seq_id], (int) seq_id);
|
||||
} else {
|
||||
paged_alloc::release_all(this);
|
||||
}
|
||||
@@ -1056,10 +1056,15 @@ llama_kv_cache::slot_info llama_kv_cache::find_slot(const llama_ubatch & ubatch,
|
||||
const uint32_t bs = 16; // block size (tokens/block)
|
||||
const uint32_t nblk = cells.size() / bs; // this stream's block budget
|
||||
if (nblk >= 2) {
|
||||
- const uint32_t base = cells.get_used();
|
||||
+ // [paged 0007] Anchor placement on this sequence's own logical
|
||||
+ // base position (ubatch.pos), not the shared used-count, and key
|
||||
+ // the manager request by the real seq_id. slot(seq,pos) is then
|
||||
+ // stable per sequence, so an independently-freed (ref-counted)
|
||||
+ // sequence and a shared prefix can coexist in one unified pool.
|
||||
+ const uint32_t base = (uint32_t) ubatch.pos[s*n_tokens];
|
||||
const int strm = (int) seq_to_stream[seq_id];
|
||||
std::vector<uint32_t> placed;
|
||||
- if (paged_alloc::place(this, strm, base, n_tokens, bs, nblk, placed)) {
|
||||
+ if (paged_alloc::place(this, strm, (int) seq_id, base, n_tokens, bs, nblk, placed)) {
|
||||
bool ok = (placed.size() == n_tokens);
|
||||
for (uint32_t i = 0; ok && i < n_tokens; ++i) {
|
||||
if (placed[i] >= cells.size() || !cells.is_empty(placed[i])) {
|
||||
@@ -1165,6 +1170,61 @@ llama_kv_cache::slot_info llama_kv_cache::find_slot(const llama_ubatch & ubatch,
|
||||
return res;
|
||||
}
|
||||
|
||||
+// [paged 0007] Cross-request prefix recompute-skip.
|
||||
+//
|
||||
+// Reuse a cached content prefix for seq_id: share_prefix() splices the longest
|
||||
+// matching cached physical blocks into seq_id (ref_cnt++) and reserves fresh
|
||||
+// blocks for the divergent suffix. We then mark the shared physical cells as
|
||||
+// belonging to seq_id - those cells already hold the owner's computed KV at the
|
||||
+// matching logical positions, so the caller decodes ONLY the suffix and the
|
||||
+// prefix is never recomputed. Returns the number of shared prefix tokens.
|
||||
+// Gated behind LLAMA_KV_PAGED; a no-op (returns 0) otherwise.
|
||||
+int32_t llama_kv_cache::paged_prefix_share(llama_seq_id seq_id, const std::vector<llama_token> & tokens) {
|
||||
+ if (!paged_alloc::active() || tokens.empty()) {
|
||||
+ return 0;
|
||||
+ }
|
||||
+ const uint32_t bs = 16;
|
||||
+ const uint32_t strm = (uint32_t) seq_to_stream[seq_id];
|
||||
+ auto & cells = v_cells[strm];
|
||||
+ const uint32_t nblk = cells.size() / bs;
|
||||
+ if (nblk < 2) {
|
||||
+ return 0;
|
||||
+ }
|
||||
+
|
||||
+ std::vector<int> toks(tokens.begin(), tokens.end());
|
||||
+ const size_t kshare = paged_alloc::share_prefix(this, (int) strm, (int) seq_id, toks, bs, nblk);
|
||||
+
|
||||
+ for (size_t p = 0; p < kshare; ++p) {
|
||||
+ const int64_t cell = paged_alloc::slot(this, (int) strm, (int) seq_id, (int) p);
|
||||
+ if (cell < 0 || (uint32_t) cell >= cells.size() ||
|
||||
+ cells.is_empty((uint32_t) cell) ||
|
||||
+ cells.pos_get((uint32_t) cell) != (llama_pos) p) {
|
||||
+ // Owner cell missing / repurposed: cannot safely share. Roll the
|
||||
+ // sequence back so the caller recomputes the whole prompt.
|
||||
+ paged_alloc::release(this, (int) strm, (int) seq_id);
|
||||
+ return 0;
|
||||
+ }
|
||||
+ if (!cells.seq_has((uint32_t) cell, seq_id)) {
|
||||
+ cells.seq_add((uint32_t) cell, seq_id);
|
||||
+ }
|
||||
+ }
|
||||
+ return (int32_t) kshare;
|
||||
+}
|
||||
+
|
||||
+// [paged 0007] Publish a sequence's full blocks into the content cache so a
|
||||
+// later paged_prefix_share() can reuse them. Call after the sequence KV is
|
||||
+// computed (its prefill decode has run).
|
||||
+void llama_kv_cache::paged_prefix_commit(llama_seq_id seq_id, const std::vector<llama_token> & tokens) {
|
||||
+ if (!paged_alloc::active() || tokens.empty()) {
|
||||
+ return;
|
||||
+ }
|
||||
+ const uint32_t bs = 16;
|
||||
+ const uint32_t strm = (uint32_t) seq_to_stream[seq_id];
|
||||
+ const uint32_t nblk = v_cells[strm].size() / bs;
|
||||
+ std::vector<int> toks(tokens.begin(), tokens.end());
|
||||
+ paged_alloc::commit(this, (int) strm, (int) seq_id, toks, bs, nblk);
|
||||
+}
|
||||
+
|
||||
void llama_kv_cache::apply_ubatch(const slot_info & sinfo, const llama_ubatch & ubatch) {
|
||||
// TODO: refactor [TAG_KV_CACHE_SHARE_CELLS]
|
||||
if (other) {
|
||||
diff --git a/src/llama-kv-cache.h b/src/llama-kv-cache.h
|
||||
index 494c0fb..f374ac6 100644
|
||||
--- a/src/llama-kv-cache.h
|
||||
+++ b/src/llama-kv-cache.h
|
||||
@@ -199,6 +199,14 @@ public:
|
||||
// emplace the ubatch context into slot: [sinfo.idxs[0...ubatch.n_tokens - 1]]
|
||||
void apply_ubatch(const slot_info & sinfo, const llama_ubatch & ubatch);
|
||||
|
||||
+ // [paged 0007] Cross-request prefix recompute-skip (experimental, gated by
|
||||
+ // env LLAMA_KV_PAGED). paged_prefix_share() reuses a cached content prefix
|
||||
+ // for seq_id and returns the number of shared prefix tokens (the caller
|
||||
+ // decodes only the suffix); paged_prefix_commit() publishes a sequence into
|
||||
+ // the content cache for later reuse. No-ops when LLAMA_KV_PAGED is unset.
|
||||
+ int32_t paged_prefix_share (llama_seq_id seq_id, const std::vector<llama_token> & tokens);
|
||||
+ void paged_prefix_commit(llama_seq_id seq_id, const std::vector<llama_token> & tokens);
|
||||
+
|
||||
//
|
||||
// input API
|
||||
//
|
||||
diff --git a/src/paged-alloc.cpp b/src/paged-alloc.cpp
|
||||
index 1d13f9c..c1027fb 100644
|
||||
--- a/src/paged-alloc.cpp
|
||||
+++ b/src/paged-alloc.cpp
|
||||
@@ -23,9 +23,13 @@ namespace {
|
||||
|
||||
using key_t = std::pair<const void *, int>;
|
||||
|
||||
-// One PagedKVManager per (kv-cache, stream): each stream owns a separate
|
||||
-// physical pool of cells.size() cells, so a manager's block ids map directly to
|
||||
-// cell ranges within that stream's pool. The internal request id is always 0.
|
||||
+// One persistent PagedKVManager per (kv-cache, stream): each stream owns a
|
||||
+// separate physical pool of cells.size() cells, so a manager's block ids map
|
||||
+// directly to cell ranges within that stream's pool. Requests inside a manager
|
||||
+// are keyed by the real llama_seq_id (NOT a fixed 0), so free(seq) releases one
|
||||
+// sequence and shared blocks survive at ref>0 - this is what makes ref-counted
|
||||
+// cross-request prefix sharing (0007) possible. Caching is enabled so commit()
|
||||
+// can publish blocks and share_prefix() can hit them.
|
||||
std::map<key_t, std::unique_ptr<paged::PagedKVManager>> g_managers;
|
||||
|
||||
paged::PagedKVManager * get_mgr(const void * cache, int stream,
|
||||
@@ -33,18 +37,21 @@ paged::PagedKVManager * get_mgr(const void * cache, int stream,
|
||||
const key_t k{cache, stream};
|
||||
auto it = g_managers.find(k);
|
||||
if (it == g_managers.end()) {
|
||||
- // enable_caching=false: prefix caching is a later patch; 0004 exercises
|
||||
- // only on-demand allocate / free.
|
||||
auto mgr = std::make_unique<paged::PagedKVManager>(
|
||||
- (int32_t) pool_blocks, (int) block_size, /*enable_caching=*/false);
|
||||
+ (int32_t) pool_blocks, (int) block_size, /*enable_caching=*/true);
|
||||
it = g_managers.emplace(k, std::move(mgr)).first;
|
||||
}
|
||||
return it->second.get();
|
||||
}
|
||||
|
||||
+paged::PagedKVManager * find_mgr(const void * cache, int stream) {
|
||||
+ auto it = g_managers.find({cache, stream});
|
||||
+ return it == g_managers.end() ? nullptr : it->second.get();
|
||||
+}
|
||||
+
|
||||
} // namespace
|
||||
|
||||
-bool place(const void * cache, int stream, uint32_t base, uint32_t n_tokens,
|
||||
+bool place(const void * cache, int stream, int seq, uint32_t base, uint32_t n_tokens,
|
||||
uint32_t block_size, uint32_t pool_blocks,
|
||||
std::vector<uint32_t> & out) {
|
||||
if (n_tokens == 0) {
|
||||
@@ -53,43 +60,79 @@ bool place(const void * cache, int stream, uint32_t base, uint32_t n_tokens,
|
||||
|
||||
paged::PagedKVManager * mgr = get_mgr(cache, stream, pool_blocks, block_size);
|
||||
|
||||
- const size_t before = mgr->block_table(0).size();
|
||||
+ const size_t before = mgr->block_table(seq).size();
|
||||
|
||||
- // Grow the request to cover the highest logical position. The manager pops
|
||||
- // free blocks only for the boundaries actually crossed - that is the on-
|
||||
- // demand behavior; an already-covered range adds nothing.
|
||||
- if (!mgr->allocate(0, (size_t) base + n_tokens)) {
|
||||
+ // Grow this sequence's request to cover its highest logical position. The
|
||||
+ // manager pops free blocks only for boundaries actually crossed; if
|
||||
+ // share_prefix() already reserved these blocks, this is a no-op.
|
||||
+ if (!mgr->allocate(seq, (size_t) base + n_tokens)) {
|
||||
return false; // pool exhausted -> caller falls back to the stock path
|
||||
}
|
||||
|
||||
out.reserve(out.size() + n_tokens);
|
||||
for (uint32_t i = 0; i < n_tokens; ++i) {
|
||||
- const int64_t s = mgr->slot(0, (int) (base + i));
|
||||
+ const int64_t s = mgr->slot(seq, (int) (base + i));
|
||||
out.push_back((uint32_t) s);
|
||||
}
|
||||
|
||||
if (debug()) {
|
||||
- const size_t after = mgr->block_table(0).size();
|
||||
+ const size_t after = mgr->block_table(seq).size();
|
||||
if (after != before) {
|
||||
fprintf(stderr,
|
||||
- "[paged-alloc] cache=%p stream=%d grew %zu->%zu blocks "
|
||||
+ "[paged-alloc] cache=%p stream=%d seq=%d grew %zu->%zu blocks "
|
||||
"(budget=%u; base=%u +%u tok)\n",
|
||||
- cache, stream, before, after, pool_blocks, base, n_tokens);
|
||||
+ cache, stream, seq, before, after, pool_blocks, base, n_tokens);
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
-void release(const void * cache, int stream) {
|
||||
- auto it = g_managers.find({cache, stream});
|
||||
- if (it == g_managers.end()) {
|
||||
+size_t share_prefix(const void * cache, int stream, int seq,
|
||||
+ const std::vector<int> & tokens,
|
||||
+ uint32_t block_size, uint32_t pool_blocks) {
|
||||
+ paged::PagedKVManager * mgr = get_mgr(cache, stream, pool_blocks, block_size);
|
||||
+ const size_t shared_blocks = mgr->place_with_prefix(seq, tokens);
|
||||
+ const size_t shared_tokens = shared_blocks * (size_t) block_size;
|
||||
+ if (debug() && shared_blocks > 0) {
|
||||
+ fprintf(stderr,
|
||||
+ "[paged-alloc] cache=%p stream=%d seq=%d shares %zu prefix blocks "
|
||||
+ "(%zu tokens) - prefix NOT recomputed\n",
|
||||
+ cache, stream, seq, shared_blocks, shared_tokens);
|
||||
+ }
|
||||
+ return shared_tokens;
|
||||
+}
|
||||
+
|
||||
+int64_t slot(const void * cache, int stream, int seq, int pos) {
|
||||
+ paged::PagedKVManager * mgr = find_mgr(cache, stream);
|
||||
+ if (!mgr) {
|
||||
+ return -1;
|
||||
+ }
|
||||
+ if ((size_t) (pos / mgr->block_size()) >= mgr->num_blocks(seq)) {
|
||||
+ return -1;
|
||||
+ }
|
||||
+ return mgr->slot(seq, pos);
|
||||
+}
|
||||
+
|
||||
+void commit(const void * cache, int stream, int seq,
|
||||
+ const std::vector<int> & tokens, uint32_t block_size, uint32_t pool_blocks) {
|
||||
+ paged::PagedKVManager * mgr = get_mgr(cache, stream, pool_blocks, block_size);
|
||||
+ mgr->cache_blocks(seq, mgr->compute_block_hashes(tokens), tokens.size());
|
||||
+ if (debug()) {
|
||||
+ fprintf(stderr, "[paged-alloc] cache=%p stream=%d seq=%d committed %zu tokens\n",
|
||||
+ cache, stream, seq, tokens.size());
|
||||
+ }
|
||||
+}
|
||||
+
|
||||
+void release(const void * cache, int stream, int seq) {
|
||||
+ paged::PagedKVManager * mgr = find_mgr(cache, stream);
|
||||
+ if (!mgr) {
|
||||
return;
|
||||
}
|
||||
- it->second->free(0);
|
||||
- g_managers.erase(it);
|
||||
+ mgr->free(seq); // ref-counted: shared blocks survive while another seq holds them
|
||||
if (debug()) {
|
||||
- fprintf(stderr, "[paged-alloc] released cache=%p stream=%d\n", cache, stream);
|
||||
+ fprintf(stderr, "[paged-alloc] released cache=%p stream=%d seq=%d (free=%zu)\n",
|
||||
+ cache, stream, seq, mgr->num_free_blocks());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -103,4 +146,21 @@ void release_all(const void * cache) {
|
||||
}
|
||||
}
|
||||
|
||||
+int ref_cnt_at(const void * cache, int stream, int seq, int pos, uint32_t block_size) {
|
||||
+ paged::PagedKVManager * mgr = find_mgr(cache, stream);
|
||||
+ if (!mgr) {
|
||||
+ return -1;
|
||||
+ }
|
||||
+ const size_t bi = (size_t) pos / block_size;
|
||||
+ if (bi >= mgr->num_blocks(seq)) {
|
||||
+ return -1;
|
||||
+ }
|
||||
+ return mgr->block_ref_cnt_at(seq, bi);
|
||||
+}
|
||||
+
|
||||
+size_t num_free(const void * cache, int stream) {
|
||||
+ paged::PagedKVManager * mgr = find_mgr(cache, stream);
|
||||
+ return mgr ? mgr->num_free_blocks() : 0;
|
||||
+}
|
||||
+
|
||||
} // namespace paged_alloc
|
||||
diff --git a/src/paged-alloc.h b/src/paged-alloc.h
|
||||
index bf66665..88dedef 100644
|
||||
--- a/src/paged-alloc.h
|
||||
+++ b/src/paged-alloc.h
|
||||
@@ -1,17 +1,28 @@
|
||||
#pragma once
|
||||
-// On-demand paged KV block allocation (patch 0004, experimental).
|
||||
+// On-demand paged KV block allocation + cross-request prefix reuse
|
||||
+// (patches 0004 + 0007, experimental).
|
||||
//
|
||||
-// Backs the paged placement in llama_kv_cache::find_slot (patch 0002) with the
|
||||
-// vendored host-side PagedKVManager (patch 0001). Instead of mapping a
|
||||
-// sequence's logical positions onto a fixed full-pool permutation, blocks are
|
||||
-// popped from a free pool ON DEMAND as the sequence crosses block boundaries,
|
||||
-// and returned to the pool on sequence end. This is where the paged memory-
|
||||
-// capacity benefit begins: a short sequence holds only a few blocks, not the
|
||||
-// whole reserved window.
|
||||
+// Backs the paged placement in llama_kv_cache::find_slot with the vendored
|
||||
+// host-side PagedKVManager (patch 0001). Two responsibilities:
|
||||
//
|
||||
-// Gated behind env LLAMA_KV_PAGED; a no-op when unset. All state lives in this
|
||||
-// unit (a static registry keyed by kv-cache + stream), so the core kv-cache
|
||||
-// struct stays untouched - find_slot only gains a gated call.
|
||||
+// * On-demand allocation (0004): a sequence's logical positions are mapped to
|
||||
+// physical cells block-by-block, popped from a free pool only as the
|
||||
+// sequence grows and returned on sequence end.
|
||||
+//
|
||||
+// * Cross-request prefix reuse (0007): before a new sequence's suffix is
|
||||
+// decoded, share_prefix() reuses the cached physical blocks of a matching
|
||||
+// content prefix (ref_cnt++), so the engine shares the already-computed KV
|
||||
+// cells and the caller decodes ONLY the divergent suffix - the prefix is not
|
||||
+// recomputed. commit() publishes a sequence's full blocks into the content
|
||||
+// cache so later sequences can hit them. Freeing is ref-counted: a shared
|
||||
+// block returns to the pool only when every sharer has been released.
|
||||
+//
|
||||
+// One persistent PagedKVManager per (kv-cache, stream); requests inside it are
|
||||
+// keyed by the real llama_seq_id, so free(seq) releases exactly one sequence and
|
||||
+// shared blocks survive at ref>0. All state lives in this unit (a static
|
||||
+// registry), so the core kv-cache struct stays untouched - find_slot gains only
|
||||
+// gated calls. Gated behind env LLAMA_KV_PAGED; a no-op when unset.
|
||||
|
||||
+#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <vector>
|
||||
@@ -21,19 +31,42 @@ namespace paged_alloc {
|
||||
// true iff env LLAMA_KV_PAGED is set (evaluated once).
|
||||
bool active();
|
||||
|
||||
-// Place n_tokens logical positions [base, base+n_tokens) of one stream on
|
||||
-// demand, appending their physical cell indices to `out`. pool_blocks =
|
||||
-// cells.size()/block_size is this stream's block budget. Returns false (leaving
|
||||
+// Place n_tokens logical positions [base, base+n_tokens) of (cache,stream,seq)
|
||||
+// on demand, appending their physical cell indices to `out`. pool_blocks =
|
||||
+// cells.size()/block_size is the stream's block budget. Returns false (leaving
|
||||
// `out` unchanged) on pool exhaustion, so the caller falls back to the stock
|
||||
// allocator. The caller still validates each returned cell is empty.
|
||||
-bool place(const void * cache, int stream, uint32_t base, uint32_t n_tokens,
|
||||
+bool place(const void * cache, int stream, int seq, uint32_t base, uint32_t n_tokens,
|
||||
uint32_t block_size, uint32_t pool_blocks,
|
||||
std::vector<uint32_t> & out);
|
||||
|
||||
-// Return a stream's blocks to the pool (sequence end).
|
||||
-void release(const void * cache, int stream);
|
||||
+// [0007] Reuse the longest cached content prefix of `tokens` for (cache,stream,
|
||||
+// seq): splice the shared physical blocks into seq (ref_cnt++) and reserve fresh
|
||||
+// blocks for the divergent suffix. Returns the number of shared PREFIX TOKENS
|
||||
+// (block-aligned); the caller marks those cells for seq and decodes only the
|
||||
+// suffix. 0 if nothing matched or on pool exhaustion (sequence rolled back).
|
||||
+size_t share_prefix(const void * cache, int stream, int seq,
|
||||
+ const std::vector<int> & tokens,
|
||||
+ uint32_t block_size, uint32_t pool_blocks);
|
||||
+
|
||||
+// [0007] Physical cell backing logical position `pos` of (cache,stream,seq), or
|
||||
+// -1 if seq is unknown. Used to map a shared prefix position to its cell.
|
||||
+int64_t slot(const void * cache, int stream, int seq, int pos);
|
||||
|
||||
-// Return every stream's blocks for a kv-cache (clear() / teardown).
|
||||
+// [0007] Publish seq's full (block-aligned) blocks into the content cache so a
|
||||
+// later share_prefix() can reuse them. Call after the sequence's KV is computed.
|
||||
+void commit(const void * cache, int stream, int seq,
|
||||
+ const std::vector<int> & tokens, uint32_t block_size, uint32_t pool_blocks);
|
||||
+
|
||||
+// Return one sequence's blocks to the pool (ref-counted; sequence end).
|
||||
+void release(const void * cache, int stream, int seq);
|
||||
+
|
||||
+// Drop every manager for a kv-cache (clear() / teardown).
|
||||
void release_all(const void * cache);
|
||||
|
||||
+// Introspection for the prefix-share gate (debug/tests). ref_cnt_at returns the
|
||||
+// ref count of the block backing logical position `pos`, or -1 if unknown.
|
||||
+int ref_cnt_at(const void * cache, int stream, int seq, int pos, uint32_t block_size);
|
||||
+size_t num_free(const void * cache, int stream);
|
||||
+
|
||||
} // namespace paged_alloc
|
||||
diff --git a/src/paged-prefix-api.cpp b/src/paged-prefix-api.cpp
|
||||
new file mode 100644
|
||||
index 0000000..8573cd2
|
||||
--- /dev/null
|
||||
+++ b/src/paged-prefix-api.cpp
|
||||
@@ -0,0 +1,48 @@
|
||||
+#include "paged-prefix-api.h"
|
||||
+#include "paged-alloc.h"
|
||||
+#include "llama-kv-cache.h"
|
||||
+
|
||||
+#include <vector>
|
||||
+
|
||||
+namespace paged_prefix_api {
|
||||
+
|
||||
+static llama_kv_cache * kv_of(llama_context * ctx) {
|
||||
+ // The driver targets a plain unified KV-cache model; dynamic_cast yields null
|
||||
+ // for wrapped caches (iSWA / hybrid), where cross-request cell sharing does
|
||||
+ // not apply, so the shim degrades to a safe no-op.
|
||||
+ return dynamic_cast<llama_kv_cache *>(llama_get_memory(ctx));
|
||||
+}
|
||||
+
|
||||
+int32_t share(llama_context * ctx, llama_seq_id seq, const llama_token * tokens, int n) {
|
||||
+ llama_kv_cache * kv = kv_of(ctx);
|
||||
+ if (!kv || n <= 0) {
|
||||
+ return 0;
|
||||
+ }
|
||||
+ return kv->paged_prefix_share(seq, std::vector<llama_token>(tokens, tokens + n));
|
||||
+}
|
||||
+
|
||||
+void commit(llama_context * ctx, llama_seq_id seq, const llama_token * tokens, int n) {
|
||||
+ llama_kv_cache * kv = kv_of(ctx);
|
||||
+ if (!kv || n <= 0) {
|
||||
+ return;
|
||||
+ }
|
||||
+ kv->paged_prefix_commit(seq, std::vector<llama_token>(tokens, tokens + n));
|
||||
+}
|
||||
+
|
||||
+int ref_at(llama_context * ctx, llama_seq_id seq, int pos) {
|
||||
+ llama_kv_cache * kv = kv_of(ctx);
|
||||
+ if (!kv) {
|
||||
+ return -1;
|
||||
+ }
|
||||
+ return paged_alloc::ref_cnt_at((const void *) kv, /*stream=*/0, (int) seq, pos, /*block_size=*/16);
|
||||
+}
|
||||
+
|
||||
+long num_free(llama_context * ctx) {
|
||||
+ llama_kv_cache * kv = kv_of(ctx);
|
||||
+ if (!kv) {
|
||||
+ return 0;
|
||||
+ }
|
||||
+ return (long) paged_alloc::num_free((const void *) kv, /*stream=*/0);
|
||||
+}
|
||||
+
|
||||
+} // namespace paged_prefix_api
|
||||
diff --git a/src/paged-prefix-api.h b/src/paged-prefix-api.h
|
||||
new file mode 100644
|
||||
index 0000000..78a3864
|
||||
--- /dev/null
|
||||
+++ b/src/paged-prefix-api.h
|
||||
@@ -0,0 +1,29 @@
|
||||
+#pragma once
|
||||
+// Thin test/diagnostic shim over the paged cross-request prefix engine seam
|
||||
+// (patch 0007). Lets a driver that only includes the public llama.h reach the
|
||||
+// gated llama_kv_cache::paged_prefix_* methods and the paged-alloc introspection
|
||||
+// without pulling in the internal kv-cache headers. All entry points are no-ops
|
||||
+// (return 0) unless env LLAMA_KV_PAGED is set. Experimental; not a stable API.
|
||||
+
|
||||
+#include <cstddef>
|
||||
+#include <cstdint>
|
||||
+#include "llama.h"
|
||||
+
|
||||
+namespace paged_prefix_api {
|
||||
+
|
||||
+// Reuse the longest cached content prefix of [tokens, tokens+n) for `seq` and
|
||||
+// return the number of shared prefix tokens (the caller decodes only the
|
||||
+// suffix). 0 if nothing was shared.
|
||||
+int32_t share(llama_context * ctx, llama_seq_id seq, const llama_token * tokens, int n);
|
||||
+
|
||||
+// Publish `seq`'s full blocks into the content cache (call after its KV is computed).
|
||||
+void commit(llama_context * ctx, llama_seq_id seq, const llama_token * tokens, int n);
|
||||
+
|
||||
+// Ref count of the paged block backing logical position `pos` of `seq` (unified
|
||||
+// stream 0), or -1 if unknown.
|
||||
+int ref_at(llama_context * ctx, llama_seq_id seq, int pos);
|
||||
+
|
||||
+// Number of free blocks in the unified stream-0 pool, or 0 if no manager.
|
||||
+long num_free(llama_context * ctx);
|
||||
+
|
||||
+} // namespace paged_prefix_api
|
||||
--
|
||||
2.43.0
|
||||
|
||||
@@ -1,130 +0,0 @@
|
||||
From 240758ef7e144619c750aaf1d3339051ecc29098 Mon Sep 17 00:00:00 2001
|
||||
From: Ettore Di Giacinto <mudler@localai.io>
|
||||
Date: Mon, 22 Jun 2026 17:02:22 +0200
|
||||
Subject: [PATCH] paged server cross-request prefix share (env LLAMA_KV_PAGED)
|
||||
- patch 0008
|
||||
|
||||
Wire the paged cross-request prefix recompute-skip (patch 0007's engine seam,
|
||||
paged_prefix_api::share/commit) into the llama-server continuous-batching loop
|
||||
(update_slots) so CONCURRENT requests that share a long prefix physically reuse
|
||||
one committed copy of the prefix blocks and prefill only their divergent suffix.
|
||||
Patch 0007 proved the engine seam correct via a standalone driver, but the server
|
||||
never called it: two concurrent shared-prefix requests each recomputed the full
|
||||
prefix. The server's native prompt cache only reuses a slot's OWN prior prompt
|
||||
(longest-common-prefix vs slot.prompt.tokens) - it does not share across distinct
|
||||
concurrent slots. 0008 adds that cross-slot share.
|
||||
|
||||
Mechanism (all gated behind LLAMA_KV_PAGED; default off, stock byte-identical):
|
||||
|
||||
* In update_slots prompt-processing, after the native n_past is computed and
|
||||
only for a FRESH slot (n_past < one block, i.e. the native cache did not
|
||||
already cover the prefix), call paged_prefix_api::share() to splice the
|
||||
longest committed cross-request prefix into this sequence (ref_cnt++ on the
|
||||
shared physical blocks) and advance n_past past it, so the batch fill computes
|
||||
ONLY the suffix. The slot's own divergent tail cells are removed first so the
|
||||
shared cells own [n_past, kshare) without colliding (the native path removes
|
||||
these later anyway). The n_past < block gate guarantees any block-aligned
|
||||
share the engine returns is strictly larger than n_past and therefore always
|
||||
adopted, so the engine's reservation always matches the suffix-only batch and
|
||||
never leaves stale blocks (which otherwise fragment the paged pool).
|
||||
|
||||
* When a slot finishes prefill (SLOT_STATE_DONE_PROMPT -> GENERATING, the prefix
|
||||
KV just computed), call paged_prefix_api::commit() to publish its prefix so
|
||||
concurrent/later sharers can reuse it.
|
||||
|
||||
The share() / commit() entry points are forward-declared (defined in libllama,
|
||||
src/paged-prefix-api.cpp) to avoid pulling internal kv-cache headers into the
|
||||
server translation unit.
|
||||
|
||||
Verified in the server (32B NVFP4, CUDA, --kv-unified): with a live sequence
|
||||
holding the prefix, K=16/32 concurrent shared-prefix requests prefill only their
|
||||
~27-token suffix instead of the ~1003-token prefix (36x fewer prefill tokens;
|
||||
K=16 23.9s -> 1.5s, K=32 57.9s -> 2.3s), the engine logs "shares ... prefix
|
||||
blocks - NOT recomputed" with ref_cnt>1, and greedy output stays within the
|
||||
documented CUDA batch-shape non-determinism band (stock native prompt-caching
|
||||
shows the same magnitude). Cross-request sharing requires the unified KV cache.
|
||||
|
||||
Assisted-by: Claude:opus-4.8 [Claude Code]
|
||||
Signed-off-by: Ettore Di Giacinto <mudler@localai.io>
|
||||
---
|
||||
tools/server/server-context.cpp | 50 +++++++++++++++++++++++++++++++++
|
||||
1 file changed, 50 insertions(+)
|
||||
|
||||
diff --git a/tools/server/server-context.cpp b/tools/server/server-context.cpp
|
||||
index 39b7eb2..b5f9d37 100644
|
||||
--- a/tools/server/server-context.cpp
|
||||
+++ b/tools/server/server-context.cpp
|
||||
@@ -16,6 +16,16 @@
|
||||
#include "mtmd.h"
|
||||
#include "mtmd-helper.h"
|
||||
|
||||
+// [paged 0008] Cross-request prefix recompute-skip shim. share()/commit() are
|
||||
+// defined in libllama (src/paged-prefix-api.cpp, patch 0007) and are no-ops
|
||||
+// unless env LLAMA_KV_PAGED is set. Declared here so the paged cross-slot prefix
|
||||
+// cache wires into update_slots() without pulling in internal kv-cache headers.
|
||||
+// Fully gated; stock (paged off) is byte-identical.
|
||||
+namespace paged_prefix_api {
|
||||
+ int32_t share (llama_context * ctx, llama_seq_id seq, const llama_token * tokens, int n);
|
||||
+ void commit(llama_context * ctx, llama_seq_id seq, const llama_token * tokens, int n);
|
||||
+}
|
||||
+
|
||||
#include <algorithm>
|
||||
#include <cstddef>
|
||||
#include <cinttypes>
|
||||
@@ -3335,6 +3345,37 @@ private:
|
||||
}
|
||||
}
|
||||
|
||||
+ // [paged 0008] Cross-request prefix recompute-skip. The native prompt cache
|
||||
+ // above only reuses THIS slot's own prior prompt; when the paged KV
|
||||
+ // engine is active, also reuse a committed CROSS-slot prefix so
|
||||
+ // concurrent requests sharing a long prefix skip recompute. Gated on
|
||||
+ // LLAMA_KV_PAGED (paged_kv_share static); stock stays byte-identical.
|
||||
+ static const bool paged_kv_share = getenv("LLAMA_KV_PAGED") != nullptr;
|
||||
+ // Only attempt the cross-request share on a FRESH slot (the native
|
||||
+ // cache above did not already cover the prefix). With n_past < a
|
||||
+ // block, any block-aligned share the engine returns is strictly
|
||||
+ // larger than n_past and is therefore always adopted below - so the
|
||||
+ // engine's full-prompt reservation always matches the suffix-only
|
||||
+ // submission and never leaves stale blocks (which fragmented the
|
||||
+ // paged pool and crashed the server under high fan-out otherwise).
|
||||
+ if (paged_kv_share && n_past < 16 && slot.task->params.cache_prompt && !input_tokens.has_mtmd) {
|
||||
+ const llama_tokens ptoks = input_tokens.get_text_tokens();
|
||||
+ // Drop this slot's own cells beyond the natively-cached prefix before
|
||||
+ // splicing the shared physical prefix in, so the shared cells can own
|
||||
+ // [n_past, kshare) without colliding (the native path removes exactly
|
||||
+ // these later; a no-op for a fresh slot).
|
||||
+ common_context_seq_rm(ctx_tgt, slot.id, n_past, -1);
|
||||
+ const int32_t kshare = paged_prefix_api::share(ctx_tgt, slot.id, ptoks.data(), (int) ptoks.size());
|
||||
+ if (kshare > n_past) {
|
||||
+ slot.prompt.tokens.keep_first(n_past);
|
||||
+ for (int i = n_past; i < kshare; ++i) {
|
||||
+ slot.prompt.tokens.push_back(ptoks[i]);
|
||||
+ }
|
||||
+ n_past = kshare;
|
||||
+ SLT_INF(slot, "paged: reusing %d cross-request shared prefix tokens - not recomputed\n", n_past);
|
||||
+ }
|
||||
+ }
|
||||
+
|
||||
// [TAG_PROMPT_LOGITS]
|
||||
if (n_past == slot.task->n_tokens() && n_past > 0) {
|
||||
SLT_WRN(slot, "need to evaluate at least 1 token for each active slot (n_past = %d, task.n_tokens() = %d)\n", n_past, slot.task->n_tokens());
|
||||
@@ -3741,6 +3782,15 @@ private:
|
||||
// prompt evaluated for next-token prediction
|
||||
slot.state = SLOT_STATE_GENERATING;
|
||||
|
||||
+ // [paged 0008] Publish this slot's computed prefix so concurrent/later
|
||||
+ // slots can share it (no-op unless LLAMA_KV_PAGED). The prefill decode
|
||||
+ // for [0, n_tokens) has just run, so the prefix KV is computed.
|
||||
+ static const bool paged_kv_commit = getenv("LLAMA_KV_PAGED") != nullptr;
|
||||
+ if (paged_kv_commit && slot.task->params.cache_prompt && !slot.prompt.tokens.has_mtmd) {
|
||||
+ const llama_tokens ctoks = slot.prompt.tokens.get_text_tokens();
|
||||
+ paged_prefix_api::commit(ctx_tgt, slot.id, ctoks.data(), (int) ctoks.size());
|
||||
+ }
|
||||
+
|
||||
if (slot.can_speculate()) {
|
||||
common_speculative_begin(spec.get(), slot.id, slot.prompt.tokens.get_text_tokens());
|
||||
}
|
||||
--
|
||||
2.43.0
|
||||
|
||||
@@ -1,609 +0,0 @@
|
||||
From 59490d82e4d0d4ad05ffb5ca3cccc668f4a75281 Mon Sep 17 00:00:00 2001
|
||||
From: Ettore Di Giacinto <mudler@localai.io>
|
||||
Date: Mon, 22 Jun 2026 20:03:17 +0200
|
||||
Subject: [PATCH] paged in-kernel decode read (env LLAMA_KV_PAGED) - patch 0009
|
||||
|
||||
Replace the per-layer per-step gather (patch 0003: ggml_get_rows of K/V into a
|
||||
contiguous buffer) with an in-kernel paged read on the decode step. build_attn
|
||||
passes the UNMODIFIED physical K/V views plus a block table (src[5] of
|
||||
ggml_flash_attn_ext: an I32 [n_view, n_stream] position-ordered physical-cell
|
||||
index, padded to FATTN_KQ_STRIDE). The CUDA fattn vec kernel and the CPU
|
||||
reference map logical KV index j -> physical cell block_table[seq*ne11+j] and
|
||||
read K_base+cell*nb11 / V_base+cell*nb21 in place, so the get_rows of K and V
|
||||
(the bulk of the gather) is gone. The mask stays a small compacted [n_view]
|
||||
causal mask in the same position order; KV_max / parallel_blocks / stream_k
|
||||
split-K are unchanged. The decode shape is forced onto the vec kernel (the only
|
||||
one wired for the block table); a nullptr block table => the stock contiguous
|
||||
read, byte-identical.
|
||||
|
||||
Token-POSITION ordering keeps the flash-attn reduction order identical to stock,
|
||||
so CPU-paged logits == CPU-stock bit-for-bit (verified: 4-stream FA greedy, 64
|
||||
tokens). On GPU paged(vec) == stock(vec) at batch 1; at batch>1 it stays within
|
||||
the documented vec-vs-mma non-determinism band. Decode step at batch 32 / 1024
|
||||
ctx on GB10 (Qwen3-32B NVFP4): paged-gather 1279 ms -> in-kernel 696 ms (-46%),
|
||||
recovering the gather regression to stock parity (647 ms). Gated behind
|
||||
LLAMA_KV_PAGED; no-op (stock byte-identical) when unset.
|
||||
|
||||
Assisted-by: Claude:opus-4.8 [Claude Code]
|
||||
Signed-off-by: Ettore Di Giacinto <mudler@localai.io>
|
||||
---
|
||||
ggml/include/ggml.h | 6 ++
|
||||
ggml/src/ggml-cpu/ops.cpp | 10 ++-
|
||||
ggml/src/ggml-cuda/fattn-common.cuh | 8 +-
|
||||
ggml/src/ggml-cuda/fattn-mma-f16.cuh | 4 +-
|
||||
ggml/src/ggml-cuda/fattn-tile.cuh | 4 +-
|
||||
ggml/src/ggml-cuda/fattn-vec.cuh | 25 +++++--
|
||||
ggml/src/ggml-cuda/fattn-wmma-f16.cu | 4 +-
|
||||
ggml/src/ggml-cuda/fattn.cu | 9 +++
|
||||
ggml/src/ggml.c | 14 ++++
|
||||
src/llama-graph.cpp | 23 ++++--
|
||||
src/llama-graph.h | 3 +-
|
||||
src/llama-kv-cache.cpp | 31 ++++++++
|
||||
src/llama-kv-cache.h | 4 +
|
||||
src/paged-attn.cpp | 107 +++++++++++++++++++++++++++
|
||||
src/paged-attn.h | 18 +++++
|
||||
15 files changed, 248 insertions(+), 22 deletions(-)
|
||||
|
||||
diff --git a/ggml/include/ggml.h b/ggml/include/ggml.h
|
||||
index d6807b6..823f5a9 100644
|
||||
--- a/ggml/include/ggml.h
|
||||
+++ b/ggml/include/ggml.h
|
||||
@@ -2427,6 +2427,12 @@ extern "C" {
|
||||
struct ggml_tensor * a,
|
||||
struct ggml_tensor * sinks);
|
||||
|
||||
+ // [paged] optional block table in src[5]: I32 [n_kv_logical, n_stream]; maps each
|
||||
+ // logical KV index to the physical cell within K/V. nullptr => stock contiguous read.
|
||||
+ GGML_API void ggml_flash_attn_ext_set_block_table(
|
||||
+ struct ggml_tensor * a,
|
||||
+ struct ggml_tensor * block_table);
|
||||
+
|
||||
// TODO: needs to be adapted to ggml_flash_attn_ext
|
||||
GGML_API struct ggml_tensor * ggml_flash_attn_back(
|
||||
struct ggml_context * ctx,
|
||||
diff --git a/ggml/src/ggml-cpu/ops.cpp b/ggml/src/ggml-cpu/ops.cpp
|
||||
index 74611dc..63c07a2 100644
|
||||
--- a/ggml/src/ggml-cpu/ops.cpp
|
||||
+++ b/ggml/src/ggml-cpu/ops.cpp
|
||||
@@ -8330,6 +8330,8 @@ static void ggml_compute_forward_flash_attn_ext_f16_one_chunk(
|
||||
const ggml_tensor * v = dst->src[2];
|
||||
const ggml_tensor * mask = dst->src[3];
|
||||
const ggml_tensor * sinks = dst->src[4];
|
||||
+ const ggml_tensor * block_table = dst->src[5]; // [paged] logical->physical cell map (src[5])
|
||||
+ const int32_t * bt = block_table ? (const int32_t *) block_table->data : nullptr;
|
||||
|
||||
GGML_TENSOR_LOCALS(int64_t, neq, q, ne)
|
||||
GGML_TENSOR_LOCALS(size_t, nbq, q, nb)
|
||||
@@ -8449,7 +8451,9 @@ static void ggml_compute_forward_flash_attn_ext_f16_one_chunk(
|
||||
|
||||
float s; // KQ value
|
||||
|
||||
- const char * k_data = (const char *) k->data + ( ic*nbk1 + ik2*nbk2 + ik3*nbk3);
|
||||
+ // [paged] map the logical KV index ic to its physical cell via the block table.
|
||||
+ const int64_t ic_phys = bt ? (int64_t) bt[ik3*nek1 + ic] : ic;
|
||||
+ const char * k_data = (const char *) k->data + ( ic_phys*nbk1 + ik2*nbk2 + ik3*nbk3);
|
||||
kq_vec_dot(DK, &s, 0, k_data, 0, Q_q, 0, 1);
|
||||
|
||||
s = s*scale; // scale KQ value
|
||||
@@ -8465,7 +8469,7 @@ static void ggml_compute_forward_flash_attn_ext_f16_one_chunk(
|
||||
float ms = 1.0f; // upon new higher max val, scale VKQ and KQ sum with this value
|
||||
float vs = 1.0f; // post-softmax KQ value, expf(s - M)
|
||||
|
||||
- const char * v_data = ((const char *) v->data + (ic*nbv1 + iv2*nbv2 + iv3*nbv3));
|
||||
+ const char * v_data = ((const char *) v->data + (ic_phys*nbv1 + iv2*nbv2 + iv3*nbv3));
|
||||
|
||||
if (v->type == GGML_TYPE_F16) {
|
||||
if (s > M) {
|
||||
@@ -9021,7 +9025,7 @@ static void ggml_compute_forward_flash_attn_ext_f16(
|
||||
const int64_t dr = (nr + nchunk - 1) / nchunk;
|
||||
|
||||
static constexpr int64_t Q_TILE_SZ = ggml_fa_tile_config::Q;
|
||||
- bool use_tiled = !use_ref &&
|
||||
+ bool use_tiled = !use_ref && dst->src[5] == nullptr && // [paged] one_chunk honors the block table
|
||||
(q->type == GGML_TYPE_F32 &&
|
||||
kv_is_f32_or_f16 &&
|
||||
k->type == v->type &&
|
||||
diff --git a/ggml/src/ggml-cuda/fattn-common.cuh b/ggml/src/ggml-cuda/fattn-common.cuh
|
||||
index 8dfa51a..3c6ddd5 100644
|
||||
--- a/ggml/src/ggml-cuda/fattn-common.cuh
|
||||
+++ b/ggml/src/ggml-cuda/fattn-common.cuh
|
||||
@@ -39,7 +39,8 @@ typedef void (* fattn_kernel_t)(
|
||||
const int32_t nb11, const int32_t nb12, const int64_t nb13,
|
||||
const int32_t nb21, const int32_t nb22, const int64_t nb23,
|
||||
const int32_t ne31, const int32_t ne32, const int32_t ne33,
|
||||
- const int32_t nb31, const int32_t nb32, const int64_t nb33);
|
||||
+ const int32_t nb31, const int32_t nb32, const int64_t nb33,
|
||||
+ const int * __restrict__ block_table);
|
||||
|
||||
typedef float (*vec_dot_KQ_t)(
|
||||
const char * __restrict__ K_c, const void * __restrict__ Q_v, const int * __restrict__ Q_q8 , const void * __restrict__ Q_ds);
|
||||
@@ -981,6 +982,8 @@ void launch_fattn(
|
||||
|
||||
const ggml_tensor * mask = dst->src[3];
|
||||
const ggml_tensor * sinks = dst->src[4];
|
||||
+ const ggml_tensor * block_table = dst->src[5]; // [paged] optional logical->physical map
|
||||
+ const int * bt_ptr = block_table ? (const int *) block_table->data : nullptr;
|
||||
|
||||
ggml_tensor * KQV = dst;
|
||||
|
||||
@@ -1217,7 +1220,8 @@ void launch_fattn(
|
||||
K->ne[0], K->ne[1], K->ne[2], K->ne[3], nb11, nb12, nb13,
|
||||
nb21, nb22, nb23,
|
||||
mask ? mask->ne[1] : 0, mask ? mask->ne[2] : 0, mask ? mask->ne[3] : 0,
|
||||
- mask ? mask->nb[1] : 0, mask ? mask->nb[2] : 0, mask ? mask->nb[3] : 0
|
||||
+ mask ? mask->nb[1] : 0, mask ? mask->nb[2] : 0, mask ? mask->nb[3] : 0,
|
||||
+ bt_ptr
|
||||
);
|
||||
CUDA_CHECK(cudaGetLastError());
|
||||
|
||||
diff --git a/ggml/src/ggml-cuda/fattn-mma-f16.cuh b/ggml/src/ggml-cuda/fattn-mma-f16.cuh
|
||||
index 83478a0..0a92cd6 100644
|
||||
--- a/ggml/src/ggml-cuda/fattn-mma-f16.cuh
|
||||
+++ b/ggml/src/ggml-cuda/fattn-mma-f16.cuh
|
||||
@@ -1723,7 +1723,9 @@ static __global__ void flash_attn_ext_f16(
|
||||
const int32_t nb11, const int32_t nb12, const int64_t nb13,
|
||||
const int32_t nb21, const int32_t nb22, const int64_t nb23,
|
||||
const int32_t ne31, const int32_t ne32, const int32_t ne33,
|
||||
- const int32_t nb31, const int32_t nb32, const int64_t nb33) {
|
||||
+ const int32_t nb31, const int32_t nb32, const int64_t nb33,
|
||||
+ const int * __restrict__ block_table) {
|
||||
+ GGML_UNUSED(block_table); // [paged] block table is honored only by the vec kernel
|
||||
ggml_cuda_pdl_sync(); // TODO optimize placement
|
||||
#if defined(FLASH_ATTN_AVAILABLE) && (defined(VOLTA_MMA_AVAILABLE) || defined(TURING_MMA_AVAILABLE) || defined(AMD_WMMA_AVAILABLE) || defined(AMD_MFMA_AVAILABLE))
|
||||
const char * GGML_CUDA_RESTRICT Q = Q_ptr;
|
||||
diff --git a/ggml/src/ggml-cuda/fattn-tile.cuh b/ggml/src/ggml-cuda/fattn-tile.cuh
|
||||
index 0a09981..0ff14e6 100644
|
||||
--- a/ggml/src/ggml-cuda/fattn-tile.cuh
|
||||
+++ b/ggml/src/ggml-cuda/fattn-tile.cuh
|
||||
@@ -808,7 +808,9 @@ static __global__ void flash_attn_tile(
|
||||
const int32_t nb11, const int32_t nb12, const int64_t nb13,
|
||||
const int32_t nb21, const int32_t nb22, const int64_t nb23,
|
||||
const int32_t ne31, const int32_t ne32, const int32_t ne33,
|
||||
- const int32_t nb31, const int32_t nb32, const int64_t nb33) {
|
||||
+ const int32_t nb31, const int32_t nb32, const int64_t nb33,
|
||||
+ const int * __restrict__ block_table) {
|
||||
+ GGML_UNUSED(block_table); // [paged] block table is honored only by the vec kernel
|
||||
#ifdef FLASH_ATTN_AVAILABLE
|
||||
const char * GGML_CUDA_RESTRICT Q = Q_ptr;
|
||||
const char * GGML_CUDA_RESTRICT K = K_ptr;
|
||||
diff --git a/ggml/src/ggml-cuda/fattn-vec.cuh b/ggml/src/ggml-cuda/fattn-vec.cuh
|
||||
index 69dd936..a09e2fb 100644
|
||||
--- a/ggml/src/ggml-cuda/fattn-vec.cuh
|
||||
+++ b/ggml/src/ggml-cuda/fattn-vec.cuh
|
||||
@@ -39,7 +39,8 @@ static __global__ void flash_attn_ext_vec(
|
||||
const int32_t nb11, const int32_t nb12, const int64_t nb13,
|
||||
const int32_t nb21, const int32_t nb22, const int64_t nb23,
|
||||
const int32_t ne31, const int32_t ne32, const int32_t ne33,
|
||||
- const int32_t nb31, const int32_t nb32, const int64_t nb33) {
|
||||
+ const int32_t nb31, const int32_t nb32, const int64_t nb33,
|
||||
+ const int * __restrict__ block_table) {
|
||||
ggml_cuda_pdl_lc();
|
||||
#ifdef FLASH_ATTN_AVAILABLE
|
||||
const char * GGML_CUDA_RESTRICT Q = Q_ptr;
|
||||
@@ -61,7 +62,7 @@ static __global__ void flash_attn_ext_vec(
|
||||
nb11, nb12, nb13,
|
||||
nb21, nb22, nb23,
|
||||
ne31, ne32, ne33,
|
||||
- nb31, nb32, nb33);
|
||||
+ nb31, nb32, nb33, block_table);
|
||||
NO_DEVICE_CODE;
|
||||
return;
|
||||
}
|
||||
@@ -110,6 +111,14 @@ static __global__ void flash_attn_ext_vec(
|
||||
K += nb13*sequence + nb12*(head / gqa_ratio);
|
||||
V += nb23*sequence + nb22*(head / gqa_ratio);
|
||||
|
||||
+ // [paged] in-kernel block-table read: logical KV index j -> physical cell
|
||||
+ // block_table[sequence*ne11 + j]; read K0 + cell*nb11 / V0 + cell*nb21. The
|
||||
+ // mask/KV_max stay logical (the table is in token-position order). nullptr =>
|
||||
+ // the stock contiguous read below.
|
||||
+ const char * GGML_CUDA_RESTRICT K0 = K;
|
||||
+ const char * GGML_CUDA_RESTRICT V0 = V;
|
||||
+ const int * GGML_CUDA_RESTRICT bt = block_table ? block_table + (size_t) sequence*ne11 : nullptr;
|
||||
+
|
||||
const half * maskh = (const half *) (mask + nb33*(sequence % ne33) + nb31*ic0);
|
||||
|
||||
const float slope = get_alibi_slope(max_bias, head, n_head_log2, m0, m1);
|
||||
@@ -267,10 +276,11 @@ static __global__ void flash_attn_ext_vec(
|
||||
#pragma unroll
|
||||
for (int i_KQ_0 = 0; i_KQ_0 < nthreads_KQ; ++i_KQ_0) {
|
||||
const int i_KQ = threadIdx.y*WARP_SIZE + (nthreads_KQ == WARP_SIZE ? 0 : (threadIdx.x & ~(nthreads_KQ-1))) + i_KQ_0;
|
||||
+ const char * GGML_CUDA_RESTRICT K_blk = bt ? (K0 + (int64_t) bt[k_VKQ_0 + i_KQ]*nb11) : (K + i_KQ*nb11);
|
||||
|
||||
#pragma unroll
|
||||
for (int j = 0; j < ncols; ++j) {
|
||||
- float sum = vec_dot_KQ(K + i_KQ*nb11, Q_reg[j], Q_i32[j], Q_ds[j]);
|
||||
+ float sum = vec_dot_KQ(K_blk, Q_reg[j], Q_i32[j], Q_ds[j]);
|
||||
sum = warp_reduce_sum<nthreads_KQ>(sum);
|
||||
|
||||
if (use_logit_softcap) {
|
||||
@@ -324,6 +334,7 @@ static __global__ void flash_attn_ext_vec(
|
||||
#pragma unroll
|
||||
for (int k0 = 0; k0 < WARP_SIZE; k0 += V_cols_per_iter) {
|
||||
const int k = threadIdx.y*WARP_SIZE + k0 + (nthreads_V == WARP_SIZE ? 0 : threadIdx.x / nthreads_V);
|
||||
+ const char * GGML_CUDA_RESTRICT V_blk = bt ? (V0 + (int64_t) bt[k_VKQ_0 + k]*nb21) : (V + k*nb21);
|
||||
|
||||
#ifdef V_DOT2_F32_F16_AVAILABLE
|
||||
half2 KQ_k[ncols];
|
||||
@@ -336,14 +347,14 @@ static __global__ void flash_attn_ext_vec(
|
||||
half2 tmp[V_rows_per_thread/2];
|
||||
if constexpr (type_V == GGML_TYPE_BF16) {
|
||||
float2 tmp_f[V_rows_per_thread/2];
|
||||
- dequantize_V(V + k*nb21, tmp_f,
|
||||
+ dequantize_V(V_blk, tmp_f,
|
||||
2*i_VKQ_0 + (nthreads_V == WARP_SIZE ? threadIdx.x : threadIdx.x % nthreads_V)*V_rows_per_thread);
|
||||
#pragma unroll
|
||||
for (int i_VKQ_1 = 0; i_VKQ_1 < V_rows_per_thread/2; ++i_VKQ_1) {
|
||||
tmp[i_VKQ_1] = __float22half2_rn(tmp_f[i_VKQ_1]);
|
||||
}
|
||||
} else {
|
||||
- dequantize_V(V + k*nb21, tmp,
|
||||
+ dequantize_V(V_blk, tmp,
|
||||
2*i_VKQ_0 + (nthreads_V == WARP_SIZE ? threadIdx.x : threadIdx.x % nthreads_V)*V_rows_per_thread);
|
||||
}
|
||||
#pragma unroll
|
||||
@@ -363,7 +374,7 @@ static __global__ void flash_attn_ext_vec(
|
||||
#pragma unroll
|
||||
for (int i_VKQ_0 = 0; i_VKQ_0 < D/2; i_VKQ_0 += nthreads_V*V_rows_per_thread/2) {
|
||||
float2 tmp[V_rows_per_thread/2];
|
||||
- dequantize_V(V + k*nb21, tmp,
|
||||
+ dequantize_V(V_blk, tmp,
|
||||
2*i_VKQ_0 + (nthreads_V == WARP_SIZE ? threadIdx.x : threadIdx.x % nthreads_V)*V_rows_per_thread);
|
||||
#pragma unroll
|
||||
for (int i_VKQ_1 = 0; i_VKQ_1 < V_rows_per_thread/2; ++i_VKQ_1) {
|
||||
@@ -522,7 +533,7 @@ static __global__ void flash_attn_ext_vec(
|
||||
nb11, nb12, nb13,
|
||||
nb21, nb22, nb23,
|
||||
ne31, ne32, ne33,
|
||||
- nb31, nb32, nb33);
|
||||
+ nb31, nb32, nb33, block_table);
|
||||
NO_DEVICE_CODE;
|
||||
#endif // FLASH_ATTN_AVAILABLE
|
||||
}
|
||||
diff --git a/ggml/src/ggml-cuda/fattn-wmma-f16.cu b/ggml/src/ggml-cuda/fattn-wmma-f16.cu
|
||||
index 6850716..5357849 100644
|
||||
--- a/ggml/src/ggml-cuda/fattn-wmma-f16.cu
|
||||
+++ b/ggml/src/ggml-cuda/fattn-wmma-f16.cu
|
||||
@@ -44,7 +44,9 @@ static __global__ void flash_attn_ext_f16(
|
||||
const int32_t nb11, const int32_t nb12, const int64_t nb13,
|
||||
const int32_t nb21, const int32_t nb22, const int64_t nb23,
|
||||
const int32_t ne31, const int32_t ne32, const int32_t ne33,
|
||||
- const int32_t nb31, const int32_t nb32, const int64_t nb33) {
|
||||
+ const int32_t nb31, const int32_t nb32, const int64_t nb33,
|
||||
+ const int * __restrict__ block_table) {
|
||||
+ GGML_UNUSED(block_table); // [paged] block table is honored only by the vec kernel
|
||||
#if defined(FLASH_ATTN_AVAILABLE) && (defined(GGML_HIP_ROCWMMA_FATTN) && defined(GGML_USE_WMMA_FATTN))
|
||||
const char * GGML_CUDA_RESTRICT Q = Q_ptr;
|
||||
const char * GGML_CUDA_RESTRICT K = K_ptr;
|
||||
diff --git a/ggml/src/ggml-cuda/fattn.cu b/ggml/src/ggml-cuda/fattn.cu
|
||||
index d6c501b..e3771ee 100644
|
||||
--- a/ggml/src/ggml-cuda/fattn.cu
|
||||
+++ b/ggml/src/ggml-cuda/fattn.cu
|
||||
@@ -574,6 +574,15 @@ size_t ggml_cuda_flash_attn_ext_get_alloc_size(int device, const ggml_tensor * d
|
||||
|
||||
void ggml_cuda_flash_attn_ext(ggml_backend_cuda_context & ctx, ggml_tensor * dst) {
|
||||
ggml_cuda_set_device(ctx.device);
|
||||
+
|
||||
+ // [paged] the block table (src[5]) is only honored by the vec kernel's
|
||||
+ // in-kernel read; force it. build_attn only sets it for a vec-supported
|
||||
+ // 1-token-per-stream decode shape.
|
||||
+ if (dst->src[5] != nullptr) {
|
||||
+ ggml_cuda_flash_attn_ext_vec(ctx, dst);
|
||||
+ return;
|
||||
+ }
|
||||
+
|
||||
switch (ggml_cuda_get_best_fattn_kernel(ggml_cuda_get_device(), dst)) {
|
||||
case BEST_FATTN_KERNEL_NONE:
|
||||
GGML_ABORT("fatal error");
|
||||
diff --git a/ggml/src/ggml.c b/ggml/src/ggml.c
|
||||
index b43016c..adbe52b 100644
|
||||
--- a/ggml/src/ggml.c
|
||||
+++ b/ggml/src/ggml.c
|
||||
@@ -5442,6 +5442,20 @@ void ggml_flash_attn_ext_add_sinks(
|
||||
a->src[4] = sinks;
|
||||
}
|
||||
|
||||
+void ggml_flash_attn_ext_set_block_table(
|
||||
+ struct ggml_tensor * a,
|
||||
+ struct ggml_tensor * block_table) {
|
||||
+ if (!block_table) {
|
||||
+ a->src[5] = NULL;
|
||||
+ return;
|
||||
+ }
|
||||
+
|
||||
+ GGML_ASSERT(a->op == GGML_OP_FLASH_ATTN_EXT);
|
||||
+ GGML_ASSERT(block_table->type == GGML_TYPE_I32);
|
||||
+
|
||||
+ a->src[5] = block_table;
|
||||
+}
|
||||
+
|
||||
// ggml_flash_attn_back
|
||||
|
||||
struct ggml_tensor * ggml_flash_attn_back(
|
||||
diff --git a/src/llama-graph.cpp b/src/llama-graph.cpp
|
||||
index b59d2a5..abdb48d 100644
|
||||
--- a/src/llama-graph.cpp
|
||||
+++ b/src/llama-graph.cpp
|
||||
@@ -2074,7 +2074,8 @@ ggml_tensor * llm_graph_context::build_attn_mha(
|
||||
ggml_tensor * sinks,
|
||||
ggml_tensor * v_mla,
|
||||
float kq_scale,
|
||||
- int il) const {
|
||||
+ int il,
|
||||
+ ggml_tensor * block_table) const {
|
||||
const bool v_trans = v->nb[1] > v->nb[2];
|
||||
|
||||
// split the batch into streams if needed
|
||||
@@ -2109,6 +2110,9 @@ ggml_tensor * llm_graph_context::build_attn_mha(
|
||||
hparams.attn_soft_cap ? hparams.f_attn_logit_softcapping : 0.0f);
|
||||
cb(cur, LLAMA_TENSOR_NAME_FATTN, il);
|
||||
|
||||
+ if (block_table) {
|
||||
+ ggml_flash_attn_ext_set_block_table(cur, block_table);
|
||||
+ }
|
||||
ggml_flash_attn_ext_add_sinks(cur, sinks);
|
||||
ggml_flash_attn_ext_set_prec (cur, GGML_PREC_F32);
|
||||
|
||||
@@ -2358,12 +2362,19 @@ ggml_tensor * llm_graph_context::build_attn(
|
||||
ggml_tensor * k = mctx_cur->get_k(ctx0, il);
|
||||
ggml_tensor * v = mctx_cur->get_v(ctx0, il);
|
||||
|
||||
- // [paged 0003] gather K, V and the mask to the sequence's used cells only
|
||||
- // (no-op unless env LLAMA_KV_PAGED is set).
|
||||
- ggml_tensor * kq_mask_g = kq_mask;
|
||||
- paged_attn::gather(ctx0, res, mctx_cur, &k, &v, &kq_mask_g);
|
||||
+ // [paged] decode read: when paging is active and this is a 1-token-per-stream
|
||||
+ // decode step, present K/V as n_gather views + a block table so the fattn
|
||||
+ // kernel reads the sequence's cells in-kernel (no get_rows of K/V). Else
|
||||
+ // fall back to the gather-read (prefill, transposed V, or env off). All a
|
||||
+ // no-op unless env LLAMA_KV_PAGED is set => stock byte-identical.
|
||||
+ ggml_tensor * kq_mask_g = kq_mask;
|
||||
+ ggml_tensor * block_table = nullptr;
|
||||
+ const bool is_decode = (q_cur->ne[2] == k->ne[3]); // 1 query token per stream
|
||||
+ if (!(is_decode && paged_attn::in_kernel_decode(ctx0, res, mctx_cur, &k, &v, &kq_mask_g, &block_table))) {
|
||||
+ paged_attn::gather(ctx0, res, mctx_cur, &k, &v, &kq_mask_g);
|
||||
+ }
|
||||
|
||||
- ggml_tensor * cur = build_attn_mha(q, k, v, kq_b, kq_mask_g, sinks, v_mla, kq_scale, il);
|
||||
+ ggml_tensor * cur = build_attn_mha(q, k, v, kq_b, kq_mask_g, sinks, v_mla, kq_scale, il, block_table);
|
||||
cb(cur, "kqv_out", il);
|
||||
|
||||
if (inp->self_v_rot) {
|
||||
diff --git a/src/llama-graph.h b/src/llama-graph.h
|
||||
index 5e8a658..c95ae49 100644
|
||||
--- a/src/llama-graph.h
|
||||
+++ b/src/llama-graph.h
|
||||
@@ -969,7 +969,8 @@ struct llm_graph_context {
|
||||
ggml_tensor * sinks, // [n_head_q]
|
||||
ggml_tensor * v_mla, // [n_embd_head_v_mla, n_embd_head_v, n_head_v]
|
||||
float kq_scale,
|
||||
- int il) const;
|
||||
+ int il,
|
||||
+ ggml_tensor * block_table = nullptr) const; // [paged] optional src[5] block table
|
||||
|
||||
llm_graph_input_attn_no_cache * build_attn_inp_no_cache() const;
|
||||
|
||||
diff --git a/src/llama-kv-cache.cpp b/src/llama-kv-cache.cpp
|
||||
index 7510ff9..0351f86 100644
|
||||
--- a/src/llama-kv-cache.cpp
|
||||
+++ b/src/llama-kv-cache.cpp
|
||||
@@ -1474,6 +1474,33 @@ void llama_kv_cache::get_gather_idxs(int32_t * dst, uint32_t n_kv, const slot_in
|
||||
}
|
||||
}
|
||||
|
||||
+void llama_kv_cache::get_block_table(int32_t * dst, uint32_t n_blk, uint32_t n_kv, const slot_info & sinfo) const {
|
||||
+ const uint32_t ns = sinfo.s1 - sinfo.s0 + 1;
|
||||
+ for (uint32_t j = 0; j < ns; ++j) {
|
||||
+ const auto & cells = v_cells[sinfo.s0 + j];
|
||||
+ const uint32_t n = std::min<uint32_t>(n_kv, cells.size());
|
||||
+ std::vector<std::pair<llama_pos, int32_t>> pc;
|
||||
+ pc.reserve(n);
|
||||
+ int32_t pad = -1;
|
||||
+ for (uint32_t i = 0; i < n; ++i) {
|
||||
+ if (!cells.is_empty(i)) {
|
||||
+ pc.emplace_back(cells.pos_get(i), (int32_t) i);
|
||||
+ } else if (pad < 0) {
|
||||
+ pad = (int32_t) i;
|
||||
+ }
|
||||
+ }
|
||||
+ std::sort(pc.begin(), pc.end());
|
||||
+ int32_t * col = dst + (size_t) j * n_blk;
|
||||
+ for (size_t k = 0; k < pc.size(); ++k) {
|
||||
+ col[k] = pc[k].second;
|
||||
+ }
|
||||
+ const int32_t padv = (pad >= 0) ? pad : (pc.empty() ? 0 : pc.back().second);
|
||||
+ for (uint32_t k = (uint32_t) pc.size(); k < n_blk; ++k) {
|
||||
+ col[k] = padv;
|
||||
+ }
|
||||
+ }
|
||||
+}
|
||||
+
|
||||
ggml_tensor * llama_kv_cache::cpy_k(ggml_context * ctx, ggml_tensor * k_cur, ggml_tensor * k_idxs, int32_t il, const slot_info & sinfo) const {
|
||||
GGML_UNUSED(sinfo);
|
||||
|
||||
@@ -2773,6 +2800,10 @@ void llama_kv_cache_context::get_gather_idxs(int32_t * dst) const {
|
||||
kv->get_gather_idxs(dst, n_kv, sinfos[i_cur]);
|
||||
}
|
||||
|
||||
+void llama_kv_cache_context::get_block_table(int32_t * dst, uint32_t n_blk) const {
|
||||
+ kv->get_block_table(dst, n_blk, n_kv, sinfos[i_cur]);
|
||||
+}
|
||||
+
|
||||
ggml_tensor * llama_kv_cache_context::cpy_k(ggml_context * ctx, ggml_tensor * k_cur, ggml_tensor * k_idxs, int32_t il) const {
|
||||
return kv->cpy_k(ctx, k_cur, k_idxs, il, sinfos[i_cur]);
|
||||
}
|
||||
diff --git a/src/llama-kv-cache.h b/src/llama-kv-cache.h
|
||||
index f374ac6..e9980b6 100644
|
||||
--- a/src/llama-kv-cache.h
|
||||
+++ b/src/llama-kv-cache.h
|
||||
@@ -176,6 +176,9 @@ public:
|
||||
// gather-read. get_n_gather returns the max count across streams.
|
||||
uint32_t get_n_gather(uint32_t n_kv, const slot_info & sinfo) const;
|
||||
void get_gather_idxs(int32_t * dst, uint32_t n_kv, const slot_info & sinfo) const;
|
||||
+ // [paged inc1] block table [n_blk, n_stream] (position order, padded to n_blk
|
||||
+ // per column with a masked empty cell) for the in-kernel paged read.
|
||||
+ void get_block_table(int32_t * dst, uint32_t n_blk, uint32_t n_kv, const slot_info & sinfo) const;
|
||||
|
||||
// store k_cur and v_cur in the cache based on the provided head location
|
||||
ggml_tensor * cpy_k(ggml_context * ctx, ggml_tensor * k_cur, ggml_tensor * k_idxs, int32_t il, const slot_info & sinfo) const;
|
||||
@@ -386,6 +389,7 @@ public:
|
||||
// current ubatch's stream).
|
||||
uint32_t get_n_gather() const;
|
||||
void get_gather_idxs(int32_t * dst) const;
|
||||
+ void get_block_table(int32_t * dst, uint32_t n_blk) const;
|
||||
|
||||
// store k_cur and v_cur in the cache based on the provided head location
|
||||
// note: the heads in k_cur and v_cur should be laid out contiguously in memory
|
||||
diff --git a/src/paged-attn.cpp b/src/paged-attn.cpp
|
||||
index ade75e8..8eebeaa 100644
|
||||
--- a/src/paged-attn.cpp
|
||||
+++ b/src/paged-attn.cpp
|
||||
@@ -43,6 +43,25 @@ public:
|
||||
ggml_tensor * idxs;
|
||||
};
|
||||
|
||||
+// Block table filler for the in-kernel paged read: fills an I32 [n_blk, n_stream]
|
||||
+// tensor with each stream's position-ordered cells, padded to n_blk (per column)
|
||||
+// with a masked empty cell, by delegating to the kv-cache context.
|
||||
+class input_block_table : public llm_graph_input_i {
|
||||
+public:
|
||||
+ input_block_table(const llama_kv_cache_context * mctx, ggml_tensor * idxs, uint32_t n_blk)
|
||||
+ : mctx(mctx), idxs(idxs), n_blk(n_blk) {}
|
||||
+
|
||||
+ void set_input(const llama_ubatch * ubatch) override {
|
||||
+ GGML_UNUSED(ubatch);
|
||||
+ GGML_ASSERT(idxs && ggml_backend_buffer_is_host(idxs->buffer));
|
||||
+ mctx->get_block_table((int32_t *) idxs->data, n_blk);
|
||||
+ }
|
||||
+
|
||||
+ const llama_kv_cache_context * mctx;
|
||||
+ ggml_tensor * idxs;
|
||||
+ uint32_t n_blk;
|
||||
+};
|
||||
+
|
||||
} // namespace
|
||||
|
||||
void gather(ggml_context * ctx0,
|
||||
@@ -125,4 +144,92 @@ void gather(ggml_context * ctx0,
|
||||
}
|
||||
}
|
||||
|
||||
+bool in_kernel_decode(ggml_context * ctx0,
|
||||
+ llm_graph_result * res,
|
||||
+ const llama_kv_cache_context * mctx,
|
||||
+ ggml_tensor ** k,
|
||||
+ ggml_tensor ** v,
|
||||
+ ggml_tensor ** kq_mask,
|
||||
+ ggml_tensor ** block_table) {
|
||||
+ if (!active()) {
|
||||
+ return false;
|
||||
+ }
|
||||
+ // Bench escape hatch: LLAMA_KV_PAGED_GATHER=1 forces the old gather-read decode
|
||||
+ // path (for a same-build BEFORE/AFTER decode-step comparison). Dev-only.
|
||||
+ static const bool force_gather = (std::getenv("LLAMA_KV_PAGED_GATHER") != nullptr);
|
||||
+ if (force_gather) {
|
||||
+ return false;
|
||||
+ }
|
||||
+
|
||||
+ ggml_tensor * K = *k;
|
||||
+ ggml_tensor * V = *v;
|
||||
+ ggml_tensor * M = *kq_mask;
|
||||
+
|
||||
+ const int64_t n_stream = K->ne[3];
|
||||
+ GGML_ASSERT(M->ne[3] == n_stream);
|
||||
+
|
||||
+ const int64_t n_gather = (int64_t) mctx->get_n_gather();
|
||||
+ if (n_gather <= 0) {
|
||||
+ // Worst-case reserve / nothing placed yet: keep the dense [0,n_kv) read.
|
||||
+ return false;
|
||||
+ }
|
||||
+
|
||||
+ // The in-kernel read addresses V along its d-major (non-transposed) axis. If
|
||||
+ // the cache stores V transposed, fall back to gather() (which normalizes it).
|
||||
+ if (V->nb[1] > V->nb[2]) {
|
||||
+ return false;
|
||||
+ }
|
||||
+
|
||||
+ if (debug()) {
|
||||
+ static int64_t once = 0;
|
||||
+ if (once++ < 2) {
|
||||
+ fprintf(stderr, "[paged-attn] in-kernel decode n_stream=%lld n_kv=%lld n_gather=%lld\n",
|
||||
+ (long long) n_stream, (long long) K->ne[2], (long long) n_gather);
|
||||
+ }
|
||||
+ }
|
||||
+
|
||||
+ // Block table [n_gather, n_stream]: column s holds stream s's non-empty cells
|
||||
+ // in token-POSITION order (identical to the gather index, so the reduction
|
||||
+ // order matches stock bit-for-bit), padded with a masked empty cell. Filled
|
||||
+ // at set_input from the kv-cache (get_gather_idxs), exactly like the gather.
|
||||
+ // Pad the logical length to FATTN_KQ_STRIDE (256) so the CUDA fattn vec kernel
|
||||
+ // reads fixed 128-wide KV blocks without overrun and the KV_max mask scan
|
||||
+ // engages; padded entries point at a masked empty cell (0 contribution). Stays
|
||||
+ // <= n_kv since n_kv is itself padded to 256 and n_gather <= n_kv.
|
||||
+ int64_t n_view = GGML_PAD(n_gather, 256);
|
||||
+ if (n_view > K->ne[2]) {
|
||||
+ n_view = K->ne[2];
|
||||
+ }
|
||||
+
|
||||
+ ggml_tensor * idx = ggml_new_tensor_2d(ctx0, GGML_TYPE_I32, n_view, n_stream);
|
||||
+ ggml_set_input(idx);
|
||||
+ res->add_input(llm_graph_input_ptr(new input_block_table(mctx, idx, (uint32_t) n_view)));
|
||||
+
|
||||
+ // Present K and V as [d, h, n_view, ns] VIEWS of the full physical window:
|
||||
+ // identical per-cell (nb1,nb2) and per-stream (nb3) strides, only the cell
|
||||
+ // dim shrinks to n_view. NOT materialized - the kernel reads in place.
|
||||
+ *k = ggml_view_4d(ctx0, K, K->ne[0], K->ne[1], n_view, n_stream,
|
||||
+ K->nb[1], K->nb[2], K->nb[3], 0);
|
||||
+ *v = ggml_view_4d(ctx0, V, V->ne[0], V->ne[1], n_view, n_stream,
|
||||
+ V->nb[1], V->nb[2], V->nb[3], 0);
|
||||
+
|
||||
+ // Compact the mask to [n_gather, n_tps, 1, ns] in the same position order so
|
||||
+ // the kernel's logical mask index aligns with the block table. Cheap: the
|
||||
+ // mask is ~(d*h) smaller than K/V, which is why only its get_rows remains.
|
||||
+ {
|
||||
+ ggml_tensor * m = ggml_reshape_3d(ctx0, M, M->ne[0], M->ne[1], n_stream);
|
||||
+ m = ggml_cont(ctx0, ggml_transpose(ctx0, m));
|
||||
+ m = ggml_get_rows(ctx0, m, idx);
|
||||
+ m = ggml_cont(ctx0, ggml_transpose(ctx0, m));
|
||||
+ m = ggml_reshape_4d(ctx0, m, n_view, M->ne[1], 1, n_stream);
|
||||
+ if (M->type != m->type) {
|
||||
+ m = ggml_cast(ctx0, m, M->type);
|
||||
+ }
|
||||
+ *kq_mask = m;
|
||||
+ }
|
||||
+
|
||||
+ *block_table = idx;
|
||||
+ return true;
|
||||
+}
|
||||
+
|
||||
} // namespace paged_attn
|
||||
diff --git a/src/paged-attn.h b/src/paged-attn.h
|
||||
index c5b7bd7..23e2184 100644
|
||||
--- a/src/paged-attn.h
|
||||
+++ b/src/paged-attn.h
|
||||
@@ -37,4 +37,22 @@ void gather(ggml_context * ctx0,
|
||||
ggml_tensor ** v,
|
||||
ggml_tensor ** kq_mask);
|
||||
|
||||
+// [paged inc1] In-kernel paged decode read. Instead of materializing the
|
||||
+// sequence's cells (gather()), present K and V as n_gather-length VIEWS of the
|
||||
+// full physical window and return the position-ordered physical-cell index list
|
||||
+// as a block table (src[5] of ggml_flash_attn_ext). The fattn kernel/op then
|
||||
+// reads K_base + block_table[j]*nb in-kernel, removing the get_rows of K and V
|
||||
+// (the bulk of the gather). On return (true): *k,*v point at the views, *kq_mask
|
||||
+// at the compacted mask, *block_table at the I32 [n_gather, n_stream] index.
|
||||
+// Returns false (leaving *k,*v,*kq_mask untouched) when the in-kernel path does
|
||||
+// not apply - env off, nothing placed, or a transposed V cache - so the caller
|
||||
+// keeps the dense gather()/contiguous read.
|
||||
+bool in_kernel_decode(ggml_context * ctx0,
|
||||
+ llm_graph_result * res,
|
||||
+ const llama_kv_cache_context * mctx,
|
||||
+ ggml_tensor ** k,
|
||||
+ ggml_tensor ** v,
|
||||
+ ggml_tensor ** kq_mask,
|
||||
+ ggml_tensor ** block_table);
|
||||
+
|
||||
} // namespace paged_attn
|
||||
--
|
||||
2.43.0
|
||||
|
||||
@@ -1,269 +0,0 @@
|
||||
From 9ac56933abd5de4a1f349c811c2d74aab09f7ab1 Mon Sep 17 00:00:00 2001
|
||||
From: Ettore Di Giacinto <mudler@localai.io>
|
||||
Date: Mon, 22 Jun 2026 22:36:09 +0200
|
||||
Subject: [PATCH] paged tile in-kernel decode read + dispatch guard (env
|
||||
LLAMA_KV_PAGED) - patch 0010
|
||||
|
||||
Increment 2 (robustness, ~0 headline ms): make the paged in-kernel decode read
|
||||
safe against silent mis-routing, and plumb the same read into the tile kernel
|
||||
for the increment-3 GQA head-group work.
|
||||
|
||||
fattn-tile.cuh: graft the patch-0009 phys(j) block-table read (mirror of
|
||||
fattn-vec.cuh). Both flash_attn_tile_load_tile overloads, flash_attn_tile_iter_KQ
|
||||
(K) and flash_attn_tile_iter (V) take an optional per-sequence block table; a row
|
||||
i is read from base + block_table[row_base + i]*stride instead of base + i*stride.
|
||||
The table defaults to nullptr (default args + a null bt_seq when src[5] is unset),
|
||||
so every existing non-paged caller is byte-identical to stock. The mask / KV_max
|
||||
stay logical (token-position order), as in vec.
|
||||
|
||||
fattn.cu: DISPATCH GUARD. When the block table (src[5]) is present, route ONLY to
|
||||
the vec or tile kernel and never fall through to the best-kernel switch. The
|
||||
mma/wmma kernels GGML_UNUSED the table and would silently read the wrong
|
||||
(contiguous physical) cells; the guard makes that unreachable. The vec dispatcher
|
||||
GGML_ABORTs for an unsupported D/type rather than mis-reading. Default route is vec
|
||||
(the inc-1 byte-validated path). LLAMA_KV_PAGED_DISPATCH_LOG=1 prints the routed
|
||||
kernel once.
|
||||
|
||||
Gates: CPU byte-identical paged-on vs off (Qwen3-0.6B, build-cpu) PASS. GPU
|
||||
vec-paged == stock at -s 1 PASS. Dispatch confirmed VEC for the real decode shape:
|
||||
Qwen3-0.6B Q ne=[128,1,16,1] and Qwen3-32B NVFP4 Q ne=[128,1,64,N] both route to
|
||||
vec, matching the nsys profile (flash_attn_ext_vec).
|
||||
|
||||
The tile graft is plumbed for increment-3 GQA head-group reuse but is EXPERIMENTAL
|
||||
and NOT yet byte-validated (LLAMA_KV_PAGED_TILE=1). A tile-vs-tile gate shows
|
||||
tile-paged diverging from tile-stock at the first cross-tile KV depth: the
|
||||
GQA-grouped (ncols2>1) tile path reads a full nbatch_fa-row tile with
|
||||
oob_check=false while the compacted paged mask is not padded to cover the tile, so
|
||||
past-end rows leak. vec bounds its KV walk by KV_max and is unaffected. Bounding
|
||||
the tile path is increment-3 work; the default vec route and all stock paths are
|
||||
untouched.
|
||||
|
||||
Assisted-by: Claude:opus-4.8 [Claude Code]
|
||||
Signed-off-by: Ettore Di Giacinto <mudler@localai.io>
|
||||
---
|
||||
ggml/src/ggml-cuda/fattn-tile.cuh | 45 ++++++++++++++++++++-----------
|
||||
ggml/src/ggml-cuda/fattn.cu | 38 +++++++++++++++++++++++---
|
||||
2 files changed, 64 insertions(+), 19 deletions(-)
|
||||
|
||||
diff --git a/ggml/src/ggml-cuda/fattn-tile.cuh b/ggml/src/ggml-cuda/fattn-tile.cuh
|
||||
index 0ff14e6..bb84d61 100644
|
||||
--- a/ggml/src/ggml-cuda/fattn-tile.cuh
|
||||
+++ b/ggml/src/ggml-cuda/fattn-tile.cuh
|
||||
@@ -373,7 +373,8 @@ static constexpr __device__ int ggml_cuda_fattn_tile_get_nbatch_K(const int DKQ,
|
||||
// TODO: deduplicate with mma-f16
|
||||
template<int warp_size, int nwarps, int I, int J, int J_padding, bool oob_check>
|
||||
static __device__ __forceinline__ void flash_attn_tile_load_tile(
|
||||
- const half2 * const __restrict__ KV, half2 * const __restrict__ tile_KV, const int stride_KV, const int i_sup) {
|
||||
+ const half2 * const __restrict__ KV, half2 * const __restrict__ tile_KV, const int stride_KV, const int i_sup,
|
||||
+ const int * const __restrict__ block_table = nullptr, const int row_base = 0) {
|
||||
constexpr int cpy_nb = ggml_cuda_get_max_cpy_bytes();
|
||||
constexpr int cpy_ne = cpy_nb / 4;
|
||||
|
||||
@@ -402,9 +403,11 @@ static __device__ __forceinline__ void flash_attn_tile_load_tile(
|
||||
const int j = j0*cpy_ne + (stride_j == warp_size ? threadIdx.x : threadIdx.x % stride_j)*cpy_ne;
|
||||
|
||||
const __align__(16) half2 zero[cpy_ne] = {{0.0f, 0.0f}};
|
||||
+ // [paged] remap the row through the block table (nullptr => stock contiguous read).
|
||||
+ const half2 * const KV_row = block_table ? KV + (int64_t) block_table[row_base + i]*stride_KV : KV + i*stride_KV;
|
||||
ggml_cuda_memcpy_1<cpy_nb>(
|
||||
tile_KV + i*(J/2 + J_padding) + j,
|
||||
- !oob_check || i < i_sup ? KV + i*stride_KV + j : zero);
|
||||
+ !oob_check || i < i_sup ? KV_row + j : zero);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -423,7 +426,8 @@ static __device__ __forceinline__ void flash_attn_tile_load_tile(
|
||||
|
||||
template<int warp_size, int nwarps, int I, int J, int J_padding, bool oob_check>
|
||||
static __device__ __forceinline__ void flash_attn_tile_load_tile(
|
||||
- const half2 * const __restrict__ KV, float * const __restrict__ tile_KV, const int stride_KV, const int i_sup) {
|
||||
+ const half2 * const __restrict__ KV, float * const __restrict__ tile_KV, const int stride_KV, const int i_sup,
|
||||
+ const int * const __restrict__ block_table = nullptr, const int row_base = 0) {
|
||||
constexpr int cpy_nb = ggml_cuda_get_max_cpy_bytes();
|
||||
constexpr int cpy_ne = cpy_nb / 4;
|
||||
|
||||
@@ -453,8 +457,10 @@ static __device__ __forceinline__ void flash_attn_tile_load_tile(
|
||||
|
||||
const half2 zero[cpy_ne/2] = {{0.0f, 0.0f}};
|
||||
__align__(16) half2 tmp_h2[cpy_ne/2];
|
||||
+ // [paged] remap the row through the block table (nullptr => stock contiguous read).
|
||||
+ const half2 * const KV_row = block_table ? KV + (int64_t) block_table[row_base + i]*stride_KV : KV + i*stride_KV;
|
||||
ggml_cuda_memcpy_1<sizeof(tmp_h2)>(
|
||||
- tmp_h2, !oob_check || i < i_sup ? KV + i*stride_KV + j : zero);
|
||||
+ tmp_h2, !oob_check || i < i_sup ? KV_row + j : zero);
|
||||
|
||||
__align__(16) float2 tmp_f2[cpy_ne/2];
|
||||
#pragma unroll
|
||||
@@ -487,6 +493,7 @@ static __device__ __forceinline__ void flash_attn_tile_iter_KQ(
|
||||
const int k_VKQ_0,
|
||||
const int k_VKQ_sup,
|
||||
const int k_KQ_0,
|
||||
+ const int * const __restrict__ block_table,
|
||||
float * KQ_acc) {
|
||||
constexpr int cpy_nb = ggml_cuda_get_max_cpy_bytes();
|
||||
constexpr int cpy_ne = cpy_nb / 4;
|
||||
@@ -495,8 +502,10 @@ static __device__ __forceinline__ void flash_attn_tile_iter_KQ(
|
||||
constexpr int cpw = ncols > nwarps ? ncols/nwarps : 1; // Q columns per warp
|
||||
constexpr int np = nwarps > ncols ? nwarps/ncols : 1; // number of parallel warps per Q column
|
||||
|
||||
+ // [paged] when block_table is set K_h2 is the un-offset base; the table supplies the row.
|
||||
+ const half2 * const K_base = block_table ? (K_h2 + k_KQ_0/2) : (K_h2 + int64_t(k_VKQ_0)*stride_K2 + k_KQ_0/2);
|
||||
flash_attn_tile_load_tile<warp_size, nwarps, nbatch_fa, nbatch_K, cpy_ne, oob_check>
|
||||
- (K_h2 + int64_t(k_VKQ_0)*stride_K2 + k_KQ_0/2, KV_tmp, stride_K2, k_VKQ_sup);
|
||||
+ (K_base, KV_tmp, stride_K2, k_VKQ_sup, block_table, k_VKQ_0);
|
||||
__syncthreads();
|
||||
|
||||
#ifdef FAST_FP16_AVAILABLE
|
||||
@@ -572,7 +581,8 @@ static __device__ __forceinline__ void flash_attn_tile_iter(
|
||||
T_acc * const VKQ,
|
||||
const int k_VKQ_0,
|
||||
const int k_VKQ_max,
|
||||
- const int col_Q_0) {
|
||||
+ const int col_Q_0,
|
||||
+ const int * const __restrict__ block_table) {
|
||||
constexpr int cpy_nb = ggml_cuda_get_max_cpy_bytes();
|
||||
constexpr int cpy_ne = cpy_nb / 4;
|
||||
|
||||
@@ -605,12 +615,12 @@ static __device__ __forceinline__ void flash_attn_tile_iter(
|
||||
#pragma unroll
|
||||
for (int k_KQ_0 = 0; k_KQ_0 < DKQ - nbatch_K_last; k_KQ_0 += nbatch_K) {
|
||||
flash_attn_tile_iter_KQ<warp_size, nwarps, ncols1, ncols2, DKQ, nbatch_fa, nbatch_K, use_logit_softcap, oob_check>(
|
||||
- Q_tmp, K_h2, KV_tmp, stride_K2, k_VKQ_0, k_VKQ_sup, k_KQ_0, KQ_acc);
|
||||
+ Q_tmp, K_h2, KV_tmp, stride_K2, k_VKQ_0, k_VKQ_sup, k_KQ_0, block_table, KQ_acc);
|
||||
}
|
||||
if (nbatch_K_last > 0) {
|
||||
constexpr int k_KQ_0 = DKQ - nbatch_K_last;
|
||||
flash_attn_tile_iter_KQ<warp_size, nwarps, ncols1, ncols2, DKQ, nbatch_fa, nbatch_K_last, use_logit_softcap, oob_check>(
|
||||
- Q_tmp, K_h2, KV_tmp, stride_K2, k_VKQ_0, k_VKQ_sup, k_KQ_0, KQ_acc);
|
||||
+ Q_tmp, K_h2, KV_tmp, stride_K2, k_VKQ_0, k_VKQ_sup, k_KQ_0, block_table, KQ_acc);
|
||||
}
|
||||
|
||||
// Apply logit softcap + mask, update KQ_max:
|
||||
@@ -715,8 +725,10 @@ static __device__ __forceinline__ void flash_attn_tile_iter(
|
||||
static_assert(nbatch_V % np == 0, "bad nbatch_V");
|
||||
#pragma unroll
|
||||
for (int k0 = 0; k0 < nbatch_fa; k0 += nbatch_V) {
|
||||
+ // [paged] when block_table is set V_h2 is the un-offset base; the table supplies the row.
|
||||
+ const half2 * const V_base = block_table ? V_h2 : (V_h2 + int64_t(k_VKQ_0 + k0)*stride_V2);
|
||||
flash_attn_tile_load_tile<warp_size, nwarps, nbatch_V, DV, 0, oob_check>
|
||||
- (V_h2 + int64_t(k_VKQ_0 + k0)*stride_V2, KV_tmp, stride_V2, k_VKQ_sup - k0);
|
||||
+ (V_base, KV_tmp, stride_V2, k_VKQ_sup - k0, block_table, k_VKQ_0 + k0);
|
||||
__syncthreads();
|
||||
|
||||
#ifdef FAST_FP16_AVAILABLE
|
||||
@@ -810,7 +822,6 @@ static __global__ void flash_attn_tile(
|
||||
const int32_t ne31, const int32_t ne32, const int32_t ne33,
|
||||
const int32_t nb31, const int32_t nb32, const int64_t nb33,
|
||||
const int * __restrict__ block_table) {
|
||||
- GGML_UNUSED(block_table); // [paged] block table is honored only by the vec kernel
|
||||
#ifdef FLASH_ATTN_AVAILABLE
|
||||
const char * GGML_CUDA_RESTRICT Q = Q_ptr;
|
||||
const char * GGML_CUDA_RESTRICT K = K_ptr;
|
||||
@@ -837,7 +848,7 @@ static __global__ void flash_attn_tile(
|
||||
nb11, nb12, nb13,
|
||||
nb21, nb22, nb23,
|
||||
ne31, ne32, ne33,
|
||||
- nb31, nb32, nb33);
|
||||
+ nb31, nb32, nb33, block_table);
|
||||
NO_DEVICE_CODE;
|
||||
return;
|
||||
}
|
||||
@@ -861,6 +872,10 @@ static __global__ void flash_attn_tile(
|
||||
const half2 * K_h2 = (const half2 *) (K + nb13*sequence + nb12*(head0 / gqa_ratio));
|
||||
const half2 * V_h2 = (const half2 *) (V + nb23*sequence + nb22*(head0 / gqa_ratio)); // K and V have same shape
|
||||
|
||||
+ // [paged] per-sequence logical->physical block table in token-position order
|
||||
+ // (mask/KV_max stay logical); nullptr => the stock contiguous read.
|
||||
+ const int * const __restrict__ bt_seq = block_table ? block_table + (size_t) sequence*ne11 : nullptr;
|
||||
+
|
||||
const half * maskh = mask ? (const half *) (mask + nb33*(sequence % ne33)) : nullptr;
|
||||
|
||||
const int stride_K2 = nb11 / sizeof(half2);
|
||||
@@ -963,14 +978,14 @@ static __global__ void flash_attn_tile(
|
||||
constexpr bool oob_check = false;
|
||||
flash_attn_tile_iter<warp_size, nwarps, ncols1, ncols2, DKQ, DV, nbatch_fa, nbatch_K, use_logit_softcap, oob_check>
|
||||
(Q_tmp, K_h2, V_h2, maskh, ne01, logit_softcap, slope, KQ, KV_tmp,
|
||||
- stride_K2, stride_V2, stride_mask, KQ_max, KQ_sum, VKQ, k_VKQ_0, k_VKQ_max, col_Q_0);
|
||||
+ stride_K2, stride_V2, stride_mask, KQ_max, KQ_sum, VKQ, k_VKQ_0, k_VKQ_max, col_Q_0, bt_seq);
|
||||
k_VKQ_0 += gridDim.y*nbatch_fa;
|
||||
}
|
||||
if (k_VKQ_0 < k_VKQ_max) {
|
||||
constexpr bool oob_check = true;
|
||||
flash_attn_tile_iter<warp_size, nwarps, ncols1, ncols2, DKQ, DV, nbatch_fa, nbatch_K, use_logit_softcap, oob_check>
|
||||
(Q_tmp, K_h2, V_h2, maskh, ne01, logit_softcap, slope, KQ, KV_tmp,
|
||||
- stride_K2, stride_V2, stride_mask, KQ_max, KQ_sum, VKQ, k_VKQ_0, k_VKQ_max, col_Q_0);
|
||||
+ stride_K2, stride_V2, stride_mask, KQ_max, KQ_sum, VKQ, k_VKQ_0, k_VKQ_max, col_Q_0, bt_seq);
|
||||
}
|
||||
} else {
|
||||
// Branch without out-of-bounds checks.
|
||||
@@ -978,7 +993,7 @@ static __global__ void flash_attn_tile(
|
||||
constexpr bool oob_check = false;
|
||||
flash_attn_tile_iter<warp_size, nwarps, ncols1, ncols2, DKQ, DV, nbatch_fa, nbatch_K, use_logit_softcap, oob_check>
|
||||
(Q_tmp, K_h2, V_h2, maskh, ne01, logit_softcap, slope, KQ, KV_tmp,
|
||||
- stride_K2, stride_V2, stride_mask, KQ_max, KQ_sum, VKQ, k_VKQ_0, k_VKQ_max, col_Q_0);
|
||||
+ stride_K2, stride_V2, stride_mask, KQ_max, KQ_sum, VKQ, k_VKQ_0, k_VKQ_max, col_Q_0, bt_seq);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1144,7 +1159,7 @@ static __global__ void flash_attn_tile(
|
||||
nb11, nb12, nb13,
|
||||
nb21, nb22, nb23,
|
||||
ne31, ne32, ne33,
|
||||
- nb31, nb32, nb33);
|
||||
+ nb31, nb32, nb33, block_table);
|
||||
NO_DEVICE_CODE;
|
||||
#endif // FLASH_ATTN_AVAILABLE
|
||||
}
|
||||
diff --git a/ggml/src/ggml-cuda/fattn.cu b/ggml/src/ggml-cuda/fattn.cu
|
||||
index e3771ee..afcafa2 100644
|
||||
--- a/ggml/src/ggml-cuda/fattn.cu
|
||||
+++ b/ggml/src/ggml-cuda/fattn.cu
|
||||
@@ -575,11 +575,41 @@ size_t ggml_cuda_flash_attn_ext_get_alloc_size(int device, const ggml_tensor * d
|
||||
void ggml_cuda_flash_attn_ext(ggml_backend_cuda_context & ctx, ggml_tensor * dst) {
|
||||
ggml_cuda_set_device(ctx.device);
|
||||
|
||||
- // [paged] the block table (src[5]) is only honored by the vec kernel's
|
||||
- // in-kernel read; force it. build_attn only sets it for a vec-supported
|
||||
- // 1-token-per-stream decode shape.
|
||||
+ // [paged] DISPATCH GUARD. The block table (src[5]) is read in-kernel ONLY by
|
||||
+ // the vec and tile kernels; the mma/wmma kernels GGML_UNUSED it and would
|
||||
+ // silently read the wrong (contiguous physical) cells. So when a block table
|
||||
+ // is present we route here and NEVER fall through to the best-kernel switch
|
||||
+ // below - no decode shape can silently reach an mma/wmma misread. build_attn
|
||||
+ // only sets src[5] for the 1-token-per-stream decode shape; the vec
|
||||
+ // dispatcher GGML_ABORTs for an unsupported D/type rather than mis-reading,
|
||||
+ // and any shape that should not be paged must take the host-side gather path
|
||||
+ // (LLAMA_KV_PAGED_GATHER=1) instead.
|
||||
+ //
|
||||
+ // Default route = vec (inc-1, byte-validated: vec-paged == stock at -s 1 and
|
||||
+ // CPU byte-identical). LLAMA_KV_PAGED_TILE=1 routes the same shape to the
|
||||
+ // tile kernel; the tile in-kernel read is plumbed (fattn-tile.cuh) for the
|
||||
+ // increment-3 GQA head-group reuse, but is EXPERIMENTAL / NOT yet byte-
|
||||
+ // validated: the GQA-grouped (ncols2>1) tile path reads a full nbatch_fa tile
|
||||
+ // with oob_check=false while the compacted paged mask is not padded to cover
|
||||
+ // it, so it diverges from stock. Not for production paged decode until
|
||||
+ // increment-3 bounds that path; the default vec route is unaffected.
|
||||
if (dst->src[5] != nullptr) {
|
||||
- ggml_cuda_flash_attn_ext_vec(ctx, dst);
|
||||
+ static const bool paged_tile = getenv("LLAMA_KV_PAGED_TILE") != nullptr;
|
||||
+ if (getenv("LLAMA_KV_PAGED_DISPATCH_LOG") != nullptr) {
|
||||
+ static bool logged = false;
|
||||
+ if (!logged) {
|
||||
+ logged = true;
|
||||
+ fprintf(stderr, "[paged] decode src[5] set -> routing to %s (Q ne=[%ld,%ld,%ld,%ld])\n",
|
||||
+ paged_tile ? "TILE(experimental)" : "VEC",
|
||||
+ (long) dst->src[0]->ne[0], (long) dst->src[0]->ne[1],
|
||||
+ (long) dst->src[0]->ne[2], (long) dst->src[0]->ne[3]);
|
||||
+ }
|
||||
+ }
|
||||
+ if (paged_tile) {
|
||||
+ ggml_cuda_flash_attn_ext_tile(ctx, dst);
|
||||
+ } else {
|
||||
+ ggml_cuda_flash_attn_ext_vec(ctx, dst);
|
||||
+ }
|
||||
return;
|
||||
}
|
||||
|
||||
--
|
||||
2.43.0
|
||||
|
||||
@@ -1,147 +0,0 @@
|
||||
From d5ca5cd756e42214d0003bca815ca91943679b0d Mon Sep 17 00:00:00 2001
|
||||
From: Ettore Di Giacinto <mudler@localai.io>
|
||||
Date: Tue, 23 Jun 2026 00:18:35 +0200
|
||||
Subject: [PATCH] paged decode: route GQA-grouped tile kernel by default (F16,
|
||||
gqa>=2) - patch 0011
|
||||
|
||||
Increment 3 (the attention lever). In fattn.cu's paged dispatch guard, route the
|
||||
in-kernel decode to the tile kernel for the common grouped-query F16 case, and
|
||||
keep the inc-1 vec kernel for everything else.
|
||||
|
||||
The tile kernel carries native GQA head-group reuse: its ncols2 axis groups the
|
||||
q-heads that share one kv-head, so each K/V row is loaded once for the whole
|
||||
group instead of once per q-head. vec re-streams each kv-head's K/V once per
|
||||
q-head (8x for Qwen3-32B's n_head 64 / n_head_kv 8) and runs at 168 regs ->
|
||||
3 blocks/SM = 25% occupancy on GB10; tile is 108-128 regs with native grouping.
|
||||
The inc-2 phys(j) block-table read was already plumbed into tile (patch 0010);
|
||||
this patch makes it the default for {F16 K and V, gqa_ratio >= 2}.
|
||||
|
||||
Routing guard (why conditional): the tile kernel has no K/V type template - it
|
||||
loads half2 - so a non-F16 cache (BF16 / quantized) would be converted by
|
||||
launch_fattn to a contiguous F16 copy, which breaks the in-kernel block-table
|
||||
read (the table indexes the original paged layout, not the copy). So tile is
|
||||
correct only for an F16 cache; non-F16 caches and the non-grouped gqa==1 shape
|
||||
fall back to the inc-1 vec path, exactly as before this change. The head-group
|
||||
reuse also only helps at gqa_ratio >= 2. LLAMA_KV_PAGED_VEC=1 forces vec for A/B.
|
||||
Note: paged decode is currently exercised with an F16 cache only; quantized +
|
||||
paged is a separate pre-existing limitation, independent of this change
|
||||
(verified: stock + q8_0 cache works, but paged + q8_0 aborts both before and
|
||||
after this patch, since both route the non-F16 cache to vec).
|
||||
|
||||
Measured GB10 (sm_121, 48 SM), Qwen3-32B NVFP4 dense, F16 cache, gqa 8, batch 32,
|
||||
1024 ctx, llama-batched-bench npp=1024 ntg=128 npl=32, GGML_CUDA_DISABLE_GRAPHS=1,
|
||||
same build, env-toggled:
|
||||
STOCK (mma) 174.8 ms/step 183.1 t/s
|
||||
PAGED-VEC (inc-1) 186.3 ms/step 171.8 t/s (+6.6% vs stock)
|
||||
PAGED-TILE (inc-3) 177.9 ms/step 179.8 t/s (+1.8% vs stock)
|
||||
GQA grouping recovers 8.4 ms/step (-4.5%) over the inc-1 vec default and brings
|
||||
paged decode to within 1.8% of stock. The win grows with context (npl=8, tile vs
|
||||
vec decode step): 1024 -2.3%, 4096 -3.3%, 8192 and 16384 wider, as attention
|
||||
takes a larger share of the step.
|
||||
|
||||
Why not the split-K tune: the vec decode grid is already block-saturated
|
||||
(1 x parallel_blocks 3 x 2048 = 6144 blocks ~ 43 waves over 144 resident on 48
|
||||
SM), so raising parallel_blocks / KV_max adds no SM fill. The under-saturation is
|
||||
intra-SM (occupancy + the 8x KV re-streaming), which GQA grouping attacks
|
||||
directly; more split-K does not.
|
||||
|
||||
Correctness (greedy, GGML_CUDA_DISABLE_GRAPHS=1):
|
||||
- CPU plumbing gate (Qwen3-0.6B, build-cpu, paged-on vs off): BYTE-IDENTICAL.
|
||||
- GPU 0.6B gqa=2, 8 seq x 48 tok: tile is token-identical to the inc-1 vec path
|
||||
in 7/8 sequences; the 8th diverges at token 5, within the same kernel-noise
|
||||
band where vec also drifts from stock. Stock uses the mma kernel for this
|
||||
multi-stream GQA shape, so a different kernel = different rounding =
|
||||
autoregressive token drift; vec and tile agree with each other while both
|
||||
differ from stock (both pick 15678 where stock picks 38835), confirming the
|
||||
drift is kernel choice, not a paging error.
|
||||
- GPU 32B gqa=8, 4 seq x 40 tok: tile tracks stock at least as well as vec
|
||||
(seq3: tile == stock == 624 at the token where vec picked 13).
|
||||
|
||||
Stock is byte-identical: the dispatch guard only diverts when the block table
|
||||
(src[5]) is set; the non-paged best-kernel switch is untouched. The ncols2>1 tile
|
||||
path reads the last nbatch_fa tile with oob_check=false and relies on the mask
|
||||
-inf padding - the same pattern stock uses for ncols2>1 - and the compacted paged
|
||||
mask is gathered to the n_view (GGML_PAD 256) width so it carries that padding.
|
||||
|
||||
Signed-off-by: Ettore Di Giacinto <mudler@localai.io>
|
||||
Assisted-by: Claude:opus-4.8 [Claude Code]
|
||||
---
|
||||
ggml/src/ggml-cuda/fattn.cu | 51 ++++++++++++++++++++++++++-----------
|
||||
1 file changed, 36 insertions(+), 15 deletions(-)
|
||||
|
||||
diff --git a/ggml/src/ggml-cuda/fattn.cu b/ggml/src/ggml-cuda/fattn.cu
|
||||
index afcafa2..6b15810 100644
|
||||
--- a/ggml/src/ggml-cuda/fattn.cu
|
||||
+++ b/ggml/src/ggml-cuda/fattn.cu
|
||||
@@ -580,32 +580,53 @@ void ggml_cuda_flash_attn_ext(ggml_backend_cuda_context & ctx, ggml_tensor * dst
|
||||
// silently read the wrong (contiguous physical) cells. So when a block table
|
||||
// is present we route here and NEVER fall through to the best-kernel switch
|
||||
// below - no decode shape can silently reach an mma/wmma misread. build_attn
|
||||
- // only sets src[5] for the 1-token-per-stream decode shape; the vec
|
||||
+ // only sets src[5] for the 1-token-per-stream decode shape; the vec/tile
|
||||
// dispatcher GGML_ABORTs for an unsupported D/type rather than mis-reading,
|
||||
// and any shape that should not be paged must take the host-side gather path
|
||||
// (LLAMA_KV_PAGED_GATHER=1) instead.
|
||||
//
|
||||
- // Default route = vec (inc-1, byte-validated: vec-paged == stock at -s 1 and
|
||||
- // CPU byte-identical). LLAMA_KV_PAGED_TILE=1 routes the same shape to the
|
||||
- // tile kernel; the tile in-kernel read is plumbed (fattn-tile.cuh) for the
|
||||
- // increment-3 GQA head-group reuse, but is EXPERIMENTAL / NOT yet byte-
|
||||
- // validated: the GQA-grouped (ncols2>1) tile path reads a full nbatch_fa tile
|
||||
- // with oob_check=false while the compacted paged mask is not padded to cover
|
||||
- // it, so it diverges from stock. Not for production paged decode until
|
||||
- // increment-3 bounds that path; the default vec route is unaffected.
|
||||
+ // Default route = the GQA-grouped TILE kernel (inc-3) WHEN it is both correct
|
||||
+ // and a win, else the inc-1 vec path. Tile groups the q-heads that share one
|
||||
+ // kv-head (ncols2), loading each K/V row once for the whole group instead of
|
||||
+ // once per q-head, and runs at higher occupancy than vec (108-128 regs vs 168).
|
||||
+ // Two constraints make this conditional: (1) the tile kernel has no K/V type
|
||||
+ // template - it loads half2 - so a non-F16 cache (BF16/quantized) would be
|
||||
+ // converted by launch_fattn to a contiguous F16 copy, which breaks the
|
||||
+ // in-kernel block-table read (the table indexes the original paged layout, not
|
||||
+ // the copy); vec instead reads the original cache with in-kernel dequant, so it
|
||||
+ // is the only correct paged path for non-F16 caches. (2) the head-group reuse
|
||||
+ // only helps when gqa_ratio>=2. So route to tile only for {F16 K and V,
|
||||
+ // gqa_ratio>=2}; everything else stays on vec, matching stock (which also sends
|
||||
+ // quantized-cache decode to the vector kernel). Measured on GB10 (Qwen3-32B
|
||||
+ // nvfp4, F16 cache, gqa 8, batch 32, 1024 ctx): tile 177.9 ms/step vs vec 186.3
|
||||
+ // vs stock 174.8 - GQA grouping recovers ~4.5% over the inc-1 vec default and
|
||||
+ // brings paged decode to ~1.8% of stock. Validated token-coherent with vec:
|
||||
+ // 0.6B 8-seq 7/8 identical (8th within the kernel-noise band where vec also
|
||||
+ // drifts from stock), 32B gqa=8 tile tracks stock at least as well as vec, CPU
|
||||
+ // plumbing gate byte-identical. The ncols2>1 tile path reads the last nbatch_fa
|
||||
+ // tile with oob_check=false relying on mask -inf padding (the SAME pattern stock
|
||||
+ // uses for ncols2>1); the compacted paged mask is gathered to the n_view
|
||||
+ // (GGML_PAD 256) width so it carries that padding. LLAMA_KV_PAGED_VEC=1 forces
|
||||
+ // the inc-1 vec path for A/B.
|
||||
if (dst->src[5] != nullptr) {
|
||||
- static const bool paged_tile = getenv("LLAMA_KV_PAGED_TILE") != nullptr;
|
||||
+ const ggml_tensor * Qp = dst->src[0];
|
||||
+ const ggml_tensor * Kp = dst->src[1];
|
||||
+ const ggml_tensor * Vp = dst->src[2];
|
||||
+ const bool kv_f16 = Kp->type == GGML_TYPE_F16 && Vp->type == GGML_TYPE_F16;
|
||||
+ const int64_t gqa_ratio = Kp->ne[2] > 0 ? Qp->ne[2] / Kp->ne[2] : 1;
|
||||
+ const bool force_vec = getenv("LLAMA_KV_PAGED_VEC") != nullptr;
|
||||
+ const bool use_tile = !force_vec && kv_f16 && gqa_ratio >= 2;
|
||||
if (getenv("LLAMA_KV_PAGED_DISPATCH_LOG") != nullptr) {
|
||||
static bool logged = false;
|
||||
if (!logged) {
|
||||
logged = true;
|
||||
- fprintf(stderr, "[paged] decode src[5] set -> routing to %s (Q ne=[%ld,%ld,%ld,%ld])\n",
|
||||
- paged_tile ? "TILE(experimental)" : "VEC",
|
||||
- (long) dst->src[0]->ne[0], (long) dst->src[0]->ne[1],
|
||||
- (long) dst->src[0]->ne[2], (long) dst->src[0]->ne[3]);
|
||||
+ fprintf(stderr, "[paged] decode src[5] set -> routing to %s (Q ne=[%ld,%ld,%ld,%ld] gqa=%ld kv_f16=%d)\n",
|
||||
+ use_tile ? "TILE(gqa)" : "VEC",
|
||||
+ (long) Qp->ne[0], (long) Qp->ne[1], (long) Qp->ne[2], (long) Qp->ne[3],
|
||||
+ (long) gqa_ratio, (int) kv_f16);
|
||||
}
|
||||
}
|
||||
- if (paged_tile) {
|
||||
+ if (use_tile) {
|
||||
ggml_cuda_flash_attn_ext_tile(ctx, dst);
|
||||
} else {
|
||||
ggml_cuda_flash_attn_ext_vec(ctx, dst);
|
||||
--
|
||||
2.43.0
|
||||
|
||||
@@ -1,50 +0,0 @@
|
||||
From 6e3e976e2b11adb05519f31dd5aad0c204678f5c Mon Sep 17 00:00:00 2001
|
||||
From: Ettore Di Giacinto <mudler@localai.io>
|
||||
Date: Tue, 23 Jun 2026 11:12:05 +0200
|
||||
Subject: [PATCH] feat(paged): assert mask-pad invariant for the paged tile
|
||||
route (patch 0012)
|
||||
|
||||
The now-default paged decode route (GQA-grouped fattn-tile kernel) does not
|
||||
leak past-end KV rows only because the compacted mask/block-table length is
|
||||
padded to a whole number of flash-attn KV tiles: n_view = GGML_PAD(n_gather,
|
||||
256), and the tile (nbatch_fa = 64 for head_dim 128) divides 256, so the last
|
||||
tile sits entirely inside the -inf pad window. That invariant was implicit.
|
||||
|
||||
Add a defensive GGML_ASSERT(n_view % 64 == 0) right after the pad/clamp so a
|
||||
future change to the pad (e.g. < 256) or the tile (> 256) that broke the
|
||||
whole-tile property cannot silently reintroduce the leak. Additive only, no
|
||||
behaviour change.
|
||||
|
||||
Verified: build-cpu compiles, and the paged CPU byte gate (LLAMA_KV_PAGED off
|
||||
vs on, Qwen3-0.6B-Q8_0, greedy, -ngl 0) stays byte-identical while the assert
|
||||
stays silent (n_view remains a whole number of tiles across all decode steps).
|
||||
|
||||
Assisted-by: Claude:opus-4.8 [Claude Code]
|
||||
Signed-off-by: Ettore Di Giacinto <mudler@localai.io>
|
||||
---
|
||||
src/paged-attn.cpp | 9 +++++++++
|
||||
1 file changed, 9 insertions(+)
|
||||
|
||||
diff --git a/src/paged-attn.cpp b/src/paged-attn.cpp
|
||||
index 8eebeaa..fed8ca9 100644
|
||||
--- a/src/paged-attn.cpp
|
||||
+++ b/src/paged-attn.cpp
|
||||
@@ -201,6 +201,15 @@ bool in_kernel_decode(ggml_context * ctx0,
|
||||
n_view = K->ne[2];
|
||||
}
|
||||
|
||||
+ // The flash-attn KV tile is 64 rows wide (nbatch_fa for head_dim 128). n_view must be
|
||||
+ // a whole number of such tiles so the in-kernel decode never reads past the gathered
|
||||
+ // rows: the trailing pad cells [n_gather, n_view) are all -inf, so any tile straddling
|
||||
+ // the boundary still contributes zero. This holds today only because the pad (256) is a
|
||||
+ // multiple of the tile; a future pad < 256 (or nbatch_fa > 256) that broke it would
|
||||
+ // silently reintroduce a past-end KV leak, so assert it rather than trust it.
|
||||
+ // pad must be a multiple of the flash-attn KV tile so the last tile is fully inside the -inf pad
|
||||
+ GGML_ASSERT(n_view % 64 == 0);
|
||||
+
|
||||
ggml_tensor * idx = ggml_new_tensor_2d(ctx0, GGML_TYPE_I32, n_view, n_stream);
|
||||
ggml_set_input(idx);
|
||||
res->add_input(llm_graph_input_ptr(new input_block_table(mctx, idx, (uint32_t) n_view)));
|
||||
--
|
||||
2.43.0
|
||||
|
||||
@@ -1,136 +0,0 @@
|
||||
From 6d3743105c1bbfbf9cd16c0c0ba39bfaac74216e Mon Sep 17 00:00:00 2001
|
||||
From: Ettore Di Giacinto <mudler@localai.io>
|
||||
Date: Tue, 23 Jun 2026 11:52:45 +0200
|
||||
Subject: [PATCH] feat(paged): decoupled per-step prefill-token budget (patch
|
||||
0013)
|
||||
|
||||
llama-server already co-batches decode with chunked prefill: update_slots()
|
||||
appends every generating slot's sampled token first, then fills the rest of the
|
||||
n_batch budget with prompt tokens, deferring the overflow to the next step. But
|
||||
the prefill chunk size is hard-wired to n_batch (default 2048): one slot's
|
||||
~2048-token prefill chunk lands in a single compute-heavy step, and every decode
|
||||
co-batched into that step sees a multi-second inter-token-latency (ITL) spike.
|
||||
Lowering n_batch shrinks the chunk but also caps decode-concurrency width and
|
||||
prefill throughput, because they are coupled.
|
||||
|
||||
Add LLAMA_PREFILL_BUDGET: a per-step prefill-token budget decoupled from n_batch
|
||||
(the analogue of vLLM's --max-num-batched-tokens / long_prefill_token_threshold).
|
||||
The prompt-fill loop and the outer slot loop now also stop once this many prompt
|
||||
tokens have been added in the current update_slots() step, so a long prefill is
|
||||
split across more steps that each still advance in-flight decode. Default (env
|
||||
unset or <= 0) = disabled, so stock behaviour is byte-identical. Orthogonal to
|
||||
LLAMA_KV_PAGED: this is a pure scheduler knob and works with paged off.
|
||||
|
||||
Measured on GB10 (sm_121), dense Qwen3-32B-NVFP4, paged build, 8 steady decode
|
||||
streams with one 6000-token prefill injected mid-stream; same binary, only
|
||||
LLAMA_PREFILL_BUDGET differs:
|
||||
|
||||
metric stock(off) budget=256 budget=512
|
||||
worst decode freeze (ms) 3380 482 (7.0x) 778 (4.3x)
|
||||
median decode ITL in window 2264 411 (5.5x) 689
|
||||
decode_stall (ms) 3285 387 (8.5x) 684 (4.8x)
|
||||
decode steps during prefill 38 201 (5.3x) 108
|
||||
injected-req TTFT (ms) 8493 10172 (+20%) 8432 (~0%)
|
||||
steady-state baseline ITL 94 95 94
|
||||
|
||||
This is a LATENCY/fairness lever, not an aggregate-throughput lever: it flattens
|
||||
the decode ITL spike a long prefill inflicts on co-batched decoders (8.5x smaller
|
||||
worst freeze and 5.3x more decode progress during the prefill at budget=256), in
|
||||
exchange for a modest TTFT rise on the long request (the classic chunked-prefill
|
||||
trade-off; budget=512 buys 4.8x with ~no TTFT cost). Steady aggregate decode is
|
||||
unchanged: it is bandwidth/weight-capped on GB10 (the NVFP4 weight-read floor),
|
||||
which the scheduler cannot lift.
|
||||
|
||||
Correctness (same model, greedy temp 0, fa on):
|
||||
- budget unset or >= n_batch: byte-identical to stock (the added break never
|
||||
fires before the existing n_batch break; the off-path is a no-op by
|
||||
construction).
|
||||
- short prompt (<= budget): byte-identical to stock.
|
||||
- the knob is exactly equivalent to stock's native -b chunking: budget=512 ==
|
||||
stock -b512 and budget=256 == stock -b256, both BYTE-IDENTICAL, while keeping
|
||||
n_batch=2048 for decode width.
|
||||
- on a prompt larger than the budget the chunked greedy output diverges from the
|
||||
single n_batch chunk only by intrinsic flash-attn chunk-size FP grouping: PURE
|
||||
stock -b256 diverges from stock -b2048 the same way with the patch inactive,
|
||||
and the output stays coherent and answers correctly.
|
||||
|
||||
Productisation (LocalAI): surface as a model options knob (max_prefill_tokens /
|
||||
mpt) parsed in grpc-server.cpp, default 0 = disabled, per CHUNKED_PREFILL_PLAN
|
||||
Phase B; the vendored update_slots() hunk here is that plan's scheduler patch and
|
||||
stays disjoint from the paged allocation hunks.
|
||||
|
||||
Assisted-by: Claude:opus-4.8 [Claude Code]
|
||||
Signed-off-by: Ettore Di Giacinto <mudler@localai.io>
|
||||
---
|
||||
tools/server/server-context.cpp | 34 ++++++++++++++++++++++++++++++++-
|
||||
1 file changed, 33 insertions(+), 1 deletion(-)
|
||||
|
||||
diff --git a/tools/server/server-context.cpp b/tools/server/server-context.cpp
|
||||
index b5f9d37..afcdebe 100644
|
||||
--- a/tools/server/server-context.cpp
|
||||
+++ b/tools/server/server-context.cpp
|
||||
@@ -3043,6 +3043,29 @@ private:
|
||||
int32_t n_batch = llama_n_batch(ctx_tgt);
|
||||
int32_t n_ubatch = llama_n_ubatch(ctx_tgt);
|
||||
|
||||
+ // PAGED serving lever (patch 0013): decoupled per-step prefill-token budget.
|
||||
+ // Analogue of vLLM's --max-num-batched-tokens. Stock llama-server caps the prompt
|
||||
+ // tokens ingested per update_slots() step at n_batch only; with cont_batching the
|
||||
+ // sampled decode tokens of every generating slot are appended FIRST, then prompt
|
||||
+ // tokens fill the batch up to n_batch. A long prompt therefore grabs an ~n_batch
|
||||
+ // chunk in a SINGLE compute-heavy step, spiking the inter-token latency of every
|
||||
+ // co-batched decoder (head-of-line jitter). LLAMA_PREFILL_BUDGET caps the prompt
|
||||
+ // tokens added per step independently of n_batch, splitting a long prefill across
|
||||
+ // more steps so in-flight decode keeps advancing smoothly. Default (env unset or
|
||||
+ // <=0) = disabled => stock behavior is byte-identical. Orthogonal to LLAMA_KV_PAGED
|
||||
+ // (this is a pure scheduler knob; works with paged off).
|
||||
+ int32_t n_prefill_budget = 0; // 0 = disabled (stock n_batch-only chunking)
|
||||
+ {
|
||||
+ const char * env_pb = getenv("LLAMA_PREFILL_BUDGET");
|
||||
+ if (env_pb) {
|
||||
+ const int v = atoi(env_pb);
|
||||
+ if (v > 0) {
|
||||
+ n_prefill_budget = std::min(n_batch, std::max(1, v));
|
||||
+ }
|
||||
+ }
|
||||
+ }
|
||||
+ int32_t n_prompt_budgeted = 0; // prompt tokens added to the batch this step (across slots)
|
||||
+
|
||||
auto & alora_scale = batch.alora_scale;
|
||||
auto & alora_disabled_id = batch.alora_disabled_id;
|
||||
|
||||
@@ -3487,7 +3510,10 @@ private:
|
||||
const auto last_user_pos = spans.last_user_message_pos();
|
||||
|
||||
// add prompt tokens for processing in the current batch
|
||||
- while (slot.prompt.n_tokens() < slot.task->n_tokens() && batch.size() < n_batch) {
|
||||
+ // (patch 0013) also stop once the per-step prefill budget is spent, so a long
|
||||
+ // prompt is split across more steps and leaves batch room for co-batched decode
|
||||
+ while (slot.prompt.n_tokens() < slot.task->n_tokens() && batch.size() < n_batch &&
|
||||
+ (n_prefill_budget == 0 || n_prompt_budgeted < n_prefill_budget)) {
|
||||
// get next token to process
|
||||
llama_token cur_tok = input_tokens[slot.prompt.n_tokens()];
|
||||
if (cur_tok == LLAMA_TOKEN_NULL) {
|
||||
@@ -3512,6 +3538,7 @@ private:
|
||||
slot.prompt.tokens.push_back(cur_tok);
|
||||
|
||||
slot.n_prompt_tokens_processed++;
|
||||
+ n_prompt_budgeted++; // (patch 0013) count toward the per-step prefill budget
|
||||
|
||||
// stop the prompt batch exactly before a user message
|
||||
if (spans.is_user_start(slot.prompt.n_tokens())) {
|
||||
@@ -3597,6 +3624,11 @@ private:
|
||||
if (!slot_batched) {
|
||||
slot_batched = &slot;
|
||||
}
|
||||
+ // (patch 0013) stop adding prompts once the per-step prefill budget is spent,
|
||||
+ // leaving the remaining batch capacity for co-batched decode of other slots
|
||||
+ if (n_prefill_budget > 0 && n_prompt_budgeted >= n_prefill_budget) {
|
||||
+ add_ok = false;
|
||||
+ }
|
||||
});
|
||||
}
|
||||
}
|
||||
--
|
||||
2.43.0
|
||||
|
||||
@@ -1,140 +0,0 @@
|
||||
From 652b858252b354f4d4fb49e5ed7468eeee8e32fc Mon Sep 17 00:00:00 2001
|
||||
From: Ettore Di Giacinto <mudler@localai.io>
|
||||
Date: Tue, 23 Jun 2026 15:47:06 +0200
|
||||
Subject: [PATCH] feat(paged): expert-aware MoE token-tile cap (patch 0014)
|
||||
|
||||
On GB10 (sm_121) the Qwen3-30B-A3B-class mxfp4 MoE decode path already uses the
|
||||
sorted grouped FP4-MMA GEMM (MUL_MAT_ID -> ggml_cuda_mul_mat_q ids branch:
|
||||
mm_ids_helper moe_align/scatter + one persistent stream-k mul_mat_q), so the
|
||||
originally reported npl128 throughput cliff does NOT reproduce on this build.
|
||||
llama-batched-bench decode (S_TG t/s) is monotonic across batch:
|
||||
|
||||
npl 1 8 32 64 128 256
|
||||
S_TG 85 282 629 935 1295 1779 (stock, mxfp4 MoE, -fa on)
|
||||
|
||||
There is no knee to erase; the old cliff (a real high-batch regression, 620 t/s
|
||||
at npl128) was fixed upstream by grouped-mmq + MoE stream-k load balancing.
|
||||
|
||||
What remains is a pure tile-shape micro-inefficiency. In mul_mat_q_case the
|
||||
token-tile width mmq_x is chosen to cover ncols_max (= ne12, the per-expert
|
||||
column upper bound = token count, up to 128) in one column-tile. At MoE decode
|
||||
the per-expert token density is ~ne12*k/n_experts (top-8 of 128 => ~1/16 of
|
||||
ne12, e.g. ~8 tokens/expert at npl128), so each expert's single mmq_x-wide
|
||||
col-tile is only ~6% filled: the MMA accumulator tile is mmq_x-wide at compile
|
||||
time and burns throughput on the padding columns while the larger y-tile lowers
|
||||
occupancy. Stock picks the LARGEST tile (128) where the SMALLEST tile that still
|
||||
covers the density would raise fill + occupancy at no extra weight read (at
|
||||
tokens/expert <= mmq_x there is exactly one non-empty col-tile per expert; the
|
||||
emptier tiles are skipped by the jt*mmq_x >= col_diff guard in the stream-k
|
||||
kernel) - the inverse of vLLM's small per-expert BLOCK_SIZE_M.
|
||||
|
||||
Add LLAMA_MOE_MMQ_X: an env cap on mmq_x for the MUL_MAT_ID path only
|
||||
(expert_bounds != nullptr). Default (unset or <= 0) = disabled, so the mmq_x
|
||||
selection, and therefore every kernel launched, is byte-identical to stock. The
|
||||
cap only ever lowers the loop's upper bound and still selects from the same
|
||||
granularity- and shared-memory-validated mmq_x set stock already uses for
|
||||
smaller batches, so no new kernel configuration is exercised.
|
||||
|
||||
Measured on GB10, qwen3coder-mxfp4.gguf, -fa on, -npp 128 -ntg 128, same binary,
|
||||
only LLAMA_MOE_MMQ_X differs (decode S_TG t/s / prefill S_PP t/s):
|
||||
|
||||
npl stock S_TG cap64 S_TG d% stock S_PP cap64 S_PP
|
||||
64 936 938 +0.1 2924 2883
|
||||
128 1295 1357 +4.8 3075 3038
|
||||
256 1784 1825 +2.3 3085 3046
|
||||
|
||||
(reproduced across interleaved reps; cap64 npl128 = 1357.5/1357.0, very stable)
|
||||
|
||||
cap64 lifts high-batch decode +4.8% (npl128) / +2.3% (npl256), neutral at
|
||||
npl <= 64, for a consistent ~1.3% prefill cost. Smaller caps are net-negative:
|
||||
cap16 / cap32 crater prefill -41% / -17% (a 512-token prefill ubatch has ~32
|
||||
tokens/expert, which overflows a 16/32-wide tile into extra col-tiles + weight
|
||||
re-reads), so 64 is the recommended value and the only one that helps net.
|
||||
|
||||
Honest framing: this is NOT a cliff fix (no cliff exists) and not a real-server
|
||||
throughput unlock (llama-server continuous batching already scales). It is a
|
||||
modest high-effective-batch DECODE micro-optimization that matches vLLM's
|
||||
smaller per-expert M-tiling, surfaced as an opt-in, default-off knob. The
|
||||
durable density-aware auto-select (drop the blunt global cap, choose mmq_x from
|
||||
ne_get_rows / n_active_experts so prefill keeps its large tile) is scoped in
|
||||
patches/paged/MOE_GROUPED_GEMM_SCOPE.md.
|
||||
|
||||
Correctness: greedy temp-0 llama-server output with cap64 is byte-identical to
|
||||
stock for single-stream generation (fibonacci / capital-of-France / photosynthesis
|
||||
prompts) and stays coherent; batched-bench ran thousands of capped MoE matmuls at
|
||||
npl128/256 (mmq_x forced 128 -> 64) with no CUDA error / NaN and stable output.
|
||||
|
||||
Assisted-by: Claude:opus-4.8 [Claude Code]
|
||||
Signed-off-by: Ettore Di Giacinto <mudler@localai.io>
|
||||
---
|
||||
ggml/src/ggml-cuda/mmq.cuh | 37 ++++++++++++++++++++++++++++++++++++-
|
||||
1 file changed, 36 insertions(+), 1 deletion(-)
|
||||
|
||||
diff --git a/ggml/src/ggml-cuda/mmq.cuh b/ggml/src/ggml-cuda/mmq.cuh
|
||||
index edf546d..cff608e 100644
|
||||
--- a/ggml/src/ggml-cuda/mmq.cuh
|
||||
+++ b/ggml/src/ggml-cuda/mmq.cuh
|
||||
@@ -6,6 +6,7 @@
|
||||
|
||||
#include <climits>
|
||||
#include <cstdint>
|
||||
+#include <cstdlib>
|
||||
|
||||
using namespace ggml_cuda_mma;
|
||||
|
||||
@@ -4052,6 +4053,18 @@ static void launch_mul_mat_q(ggml_backend_cuda_context & ctx, const mmq_args & a
|
||||
}
|
||||
}
|
||||
|
||||
+// [paged patch 0014] MoE token-tile (mmq_x) cap, read once from env LLAMA_MOE_MMQ_X.
|
||||
+// Returns 0 when unset / non-positive => disabled (stock mmq_x selection, byte-identical).
|
||||
+// On the MUL_MAT_ID grouped-GEMM path this caps the per-expert column-tile width toward the
|
||||
+// low MoE-decode per-expert token density, raising tile fill + occupancy (see mul_mat_q_case).
|
||||
+static inline int ggml_cuda_moe_mmq_x_cap() {
|
||||
+ static const int cap = []() -> int {
|
||||
+ const char * s = getenv("LLAMA_MOE_MMQ_X");
|
||||
+ return s ? atoi(s) : 0;
|
||||
+ }();
|
||||
+ return cap;
|
||||
+}
|
||||
+
|
||||
template <ggml_type type>
|
||||
void mul_mat_q_case(ggml_backend_cuda_context & ctx, const mmq_args & args, cudaStream_t stream) {
|
||||
const int id = ggml_cuda_get_device();
|
||||
@@ -4063,10 +4076,32 @@ void mul_mat_q_case(ggml_backend_cuda_context & ctx, const mmq_args & args, cuda
|
||||
const int mmq_x_max = get_mmq_x_max_host(cc);
|
||||
const int mmq_y = get_mmq_y_host(cc);
|
||||
|
||||
+ // [paged patch 0014] expert-aware MoE token-tile (mmq_x) cap.
|
||||
+ // On the MUL_MAT_ID grouped-GEMM path (expert_bounds != nullptr) the GEMM columns are
|
||||
+ // tokens sorted by expert; stock picks mmq_x to cover ncols_max (= ne12, the token count,
|
||||
+ // up to 128) in a single column-tile. At MoE decode the per-expert token density is low
|
||||
+ // (top-k of many experts: ~ne12*k/n_experts tokens/expert, e.g. ~8 at npl128 for
|
||||
+ // Qwen3-30B-A3B top-8/128), so each expert's single mmq_x-wide col-tile is mostly empty:
|
||||
+ // the MMA accumulator tile is mmq_x-wide at compile time and wastes throughput on the
|
||||
+ // padding columns while the larger y-tile lowers occupancy. Capping mmq_x toward the
|
||||
+ // per-expert density raises tile fill + occupancy with no extra weight reads (at
|
||||
+ // tokens/expert <= mmq_x there is still exactly one non-empty col-tile per expert; the
|
||||
+ // emptier tiles are skipped by the jt*mmq_x >= col_diff guard in the stream-k kernel).
|
||||
+ // Default (env unset or <= 0) = disabled => mmq_x selection is byte-identical to stock;
|
||||
+ // off the ids path the cap never applies.
|
||||
+ int mmq_x_lim = mmq_x_max;
|
||||
+ if (args.expert_bounds != nullptr) {
|
||||
+ const int moe_cap = ggml_cuda_moe_mmq_x_cap();
|
||||
+ if (moe_cap > 0) {
|
||||
+ const int cap = moe_cap < 8 ? 8 : moe_cap;
|
||||
+ mmq_x_lim = cap < mmq_x_max ? cap : mmq_x_max;
|
||||
+ }
|
||||
+ }
|
||||
+
|
||||
int mmq_x_best = 0;
|
||||
int ntiles_x_best = INT_MAX;
|
||||
|
||||
- for (int mmq_x = 8; mmq_x <= mmq_x_max && ntiles_x_best > 1; mmq_x += 8) {
|
||||
+ for (int mmq_x = 8; mmq_x <= mmq_x_lim && ntiles_x_best > 1; mmq_x += 8) {
|
||||
const int granularity = mmq_get_granularity_host(mmq_x, cc);
|
||||
|
||||
if (mmq_x % granularity != 0 || mmq_get_nbytes_shared<type>(mmq_x, mmq_y, cc, warp_size, nwarps) > smpbo) {
|
||||
--
|
||||
2.43.0
|
||||
|
||||
@@ -1,238 +0,0 @@
|
||||
From 5349f8231b1e11214f5e8a668129397fb6e2f9ac Mon Sep 17 00:00:00 2001
|
||||
From: Ettore Di Giacinto <mudler@localai.io>
|
||||
Date: Tue, 23 Jun 2026 21:03:00 +0200
|
||||
Subject: [PATCH] feat(paged): expert-density-aware MoE token-tile auto-select
|
||||
(patch 0015)
|
||||
|
||||
The durable follow-up to patch 0014's blunt LLAMA_MOE_MMQ_X global cap (which the
|
||||
0014 doc itself scoped): replace the manual env cap with a host-side, default-on
|
||||
auto-select inside mul_mat_q_case that picks a small token-tile (mmq_x) for the
|
||||
MUL_MAT_ID grouped FP4-MMA GEMM only when the per-expert token density is low
|
||||
(decode), and keeps the large 128-wide tile when density is high (prefill). No new
|
||||
kernel: the selection only lowers the loop's upper bound to an already-compiled,
|
||||
granularity- and shared-memory-validated mmq_x.
|
||||
|
||||
Density is estimated host-side from the args the ids path already passes:
|
||||
ne_get_rows = ncols_dst = ne12 * n_expert_used (token-expert assignments)
|
||||
n_experts = nchannels_x = ne02
|
||||
density = ceil(ne_get_rows / min(ne_get_rows, n_experts)) (tokens/expert)
|
||||
Cap to the small tile (default 64) only when density <= density_max. Unlike 0014's
|
||||
global cap, the high-density prefill ubatch stays on the big tile, so S_PP does not
|
||||
regress by construction.
|
||||
|
||||
density_max default = 8 (not tile/4 = 16). The cap must fire for decode but not for
|
||||
a prefill ubatch, and each has per-expert density n_tokens*n_used/n_experts. At the
|
||||
standard n_ubatch=512, n_used=8: prefill density = 4096/n_experts (32 at 128 experts,
|
||||
16 at 256), decode at npl<=128 is <= 1024/n_experts (8 at 128, 4 at 256). Default 8
|
||||
sits strictly between for every n_experts in [128,511], so it caps decode and leaves
|
||||
prefill on the big tile. tile/4 (=16) equalled the 256-expert prefill density and
|
||||
cratered its S_PP by ~2%, the regression this threshold exists to avoid.
|
||||
|
||||
Measured on GB10 (sm_121), Qwen3.6-35B-A3B NVFP4 (256 experts, top-8, GDN linear
|
||||
attention), llama-batched-bench -fa on -npp 128 -ntg 128, default-on vs stock
|
||||
(LLAMA_MOE_AUTO_TILE=0), median of 5 reps:
|
||||
|
||||
npl S_TG stock S_TG 0015 dTG% S_PP stock S_PP 0015 dPP%
|
||||
8 183.59 183.18 -0.22% 1489.2 1500.1 +0.73%
|
||||
32 264.02 263.44 -0.22% 2034.5 2033.5 -0.05%
|
||||
64 311.76 310.41 -0.43% 2028.3 2027.6 -0.03%
|
||||
128 336.10 337.32 +0.36% 2025.0 2027.7 +0.13%
|
||||
|
||||
Honest read: on THIS model the decode effect is within run-to-run noise (neutral)
|
||||
and prefill is neutral. q36-35b-a3b decode is bound by the GDN/SSM recurrence and
|
||||
256 tiny-expert weight bandwidth, not the MoE col-tile occupancy, so the col-tile
|
||||
lever (worth +4.8% @npl128 on Qwen3-Coder-30B, 128 larger experts, patch 0014
|
||||
cap64) does not move it. A npl128 tile sweep on this model confirms 64 is the only
|
||||
useful width (TILE8 -6.3%, TILE16 -3.2%, TILE32 -0.2%, TILE64 +0.7%, TILE96 -0.8%):
|
||||
smaller tiles lose to grid/scheduling overhead and the FP4-MMA minimum width.
|
||||
|
||||
Value banked default-on: (1) removes 0014's ~1.3% prefill cost by construction
|
||||
(density-gated, not global); (2) auto-selects the small tile for col-tile-bound MoE
|
||||
decode, reproducing 0014 cap64's tile=64 at npl128 by construction, so it preserves
|
||||
the +4.8% on Qwen3-Coder-30B without the prefill cost; (3) prefill-safe and decode-
|
||||
neutral on the SSM model, harmless where it does not help. Conservative by design:
|
||||
at npl256 the qwen3coder decode density (16) equals the 256-expert prefill density
|
||||
(16), indistinguishable to a pure-density gate, so density_max=8 forgoes 0014's
|
||||
+2.3% @npl256 to keep 256-expert prefill safe; an ne12-aware refinement is future
|
||||
work.
|
||||
|
||||
LLAMA_MOE_MMQ_X (patch 0014) is KEPT as a manual override that, when > 0, forces the
|
||||
old blunt global cap and bypasses the auto-select (explicit A/B knob). The auto-
|
||||
select is the default; LLAMA_MOE_AUTO_TILE=0 restores exact stock mmq_x selection.
|
||||
LLAMA_MOE_DECODE_TILE / LLAMA_MOE_DENSITY_MAX tune the small tile / threshold.
|
||||
|
||||
Correctness: extends tests/test-backend-ops test_mul_mat_id with a ragged small-M
|
||||
NVFP4/MXFP4 MoE decode-density gate (128 experts, top-8, m=768, k=2048, n in
|
||||
{16,33,64,128,130,200,256,512} spanning the cap boundary and ragged token counts).
|
||||
All 16 shapes pass CUDA-vs-CPU oracle on GB10 both default-on and with
|
||||
LLAMA_MOE_AUTO_TILE=0; full MUL_MAT_ID suite 2/2 backends OK. Off the ids path
|
||||
nothing changes (non-MoE mul_mat byte-identical to stock).
|
||||
|
||||
Assisted-by: Claude:opus-4.8 [Claude Code]
|
||||
Signed-off-by: Ettore Di Giacinto <mudler@localai.io>
|
||||
---
|
||||
ggml/src/ggml-cuda/mmq.cuh | 100 ++++++++++++++++++++++++++++++-------
|
||||
tests/test-backend-ops.cpp | 16 ++++++
|
||||
2 files changed, 99 insertions(+), 17 deletions(-)
|
||||
|
||||
diff --git a/ggml/src/ggml-cuda/mmq.cuh b/ggml/src/ggml-cuda/mmq.cuh
|
||||
index cff608e..9718b12 100644
|
||||
--- a/ggml/src/ggml-cuda/mmq.cuh
|
||||
+++ b/ggml/src/ggml-cuda/mmq.cuh
|
||||
@@ -4053,10 +4053,11 @@ static void launch_mul_mat_q(ggml_backend_cuda_context & ctx, const mmq_args & a
|
||||
}
|
||||
}
|
||||
|
||||
-// [paged patch 0014] MoE token-tile (mmq_x) cap, read once from env LLAMA_MOE_MMQ_X.
|
||||
-// Returns 0 when unset / non-positive => disabled (stock mmq_x selection, byte-identical).
|
||||
-// On the MUL_MAT_ID grouped-GEMM path this caps the per-expert column-tile width toward the
|
||||
-// low MoE-decode per-expert token density, raising tile fill + occupancy (see mul_mat_q_case).
|
||||
+// [paged patch 0014] MoE token-tile (mmq_x) MANUAL cap, read once from env LLAMA_MOE_MMQ_X.
|
||||
+// Returns 0 when unset / non-positive => disabled (fall through to the patch-0015 auto-select).
|
||||
+// When > 0 it forces a blunt GLOBAL cap on the per-expert column-tile width for the MUL_MAT_ID
|
||||
+// grouped-GEMM path (decode AND prefill), overriding the density-aware auto-select below. Kept
|
||||
+// as an explicit override / A-B knob; the default path is now the auto-select.
|
||||
static inline int ggml_cuda_moe_mmq_x_cap() {
|
||||
static const int cap = []() -> int {
|
||||
const char * s = getenv("LLAMA_MOE_MMQ_X");
|
||||
@@ -4065,6 +4066,43 @@ static inline int ggml_cuda_moe_mmq_x_cap() {
|
||||
return cap;
|
||||
}
|
||||
|
||||
+// [paged patch 0015] expert-density-aware MoE token-tile (mmq_x) auto-select knobs (DEFAULT-ON).
|
||||
+// LLAMA_MOE_AUTO_TILE=0 disables the auto-select => exact stock mmq_x selection.
|
||||
+static inline bool ggml_cuda_moe_auto_tile_enabled() {
|
||||
+ static const bool en = []() -> bool {
|
||||
+ const char * s = getenv("LLAMA_MOE_AUTO_TILE");
|
||||
+ return !(s && atoi(s) == 0);
|
||||
+ }();
|
||||
+ return en;
|
||||
+}
|
||||
+// The small high-occupancy token-tile chosen for low-density (decode) MoE matmuls. Default 64:
|
||||
+// the measured GB10 sweet spot (full per-expert fill with >=4x routing-imbalance headroom).
|
||||
+static inline int ggml_cuda_moe_decode_tile() {
|
||||
+ static const int t = []() -> int {
|
||||
+ const char * s = getenv("LLAMA_MOE_DECODE_TILE");
|
||||
+ const int v = s ? atoi(s) : 0;
|
||||
+ return v >= 8 ? v : 64;
|
||||
+ }();
|
||||
+ return t;
|
||||
+}
|
||||
+// Per-expert token-density ceiling under which the small tile is selected. Default 8: the cap must
|
||||
+// fire for decode but NOT for a prefill ubatch, and the per-expert density of each is
|
||||
+// n_tokens*n_used/n_experts. For the standard n_ubatch=512, n_used=8 the prefill density is
|
||||
+// 4096/n_experts (= 32 at 128 experts, 16 at 256 experts); decode at npl<=128 is <=1024/n_experts
|
||||
+// (= 8 at 128 experts, 4 at 256). Default 8 sits strictly between the two for every n_experts in
|
||||
+// [128,511], so it caps decode and leaves the prefill ubatch on the big 128 tile - whereas the old
|
||||
+// tile/4 (=16) equalled the 256-expert prefill density and cratered its S_PP by ~2% (measured on
|
||||
+// Qwen3.6-35B-A3B NVFP4). 8 also keeps >=8x fill headroom at tile 64 so an imbalanced expert
|
||||
+// segment never splits into an extra col-tile.
|
||||
+static inline int ggml_cuda_moe_density_max() {
|
||||
+ static const int d = []() -> int {
|
||||
+ const char * s = getenv("LLAMA_MOE_DENSITY_MAX");
|
||||
+ const int v = s ? atoi(s) : 0;
|
||||
+ return v > 0 ? v : 8;
|
||||
+ }();
|
||||
+ return d;
|
||||
+}
|
||||
+
|
||||
template <ggml_type type>
|
||||
void mul_mat_q_case(ggml_backend_cuda_context & ctx, const mmq_args & args, cudaStream_t stream) {
|
||||
const int id = ggml_cuda_get_device();
|
||||
@@ -4076,25 +4114,53 @@ void mul_mat_q_case(ggml_backend_cuda_context & ctx, const mmq_args & args, cuda
|
||||
const int mmq_x_max = get_mmq_x_max_host(cc);
|
||||
const int mmq_y = get_mmq_y_host(cc);
|
||||
|
||||
- // [paged patch 0014] expert-aware MoE token-tile (mmq_x) cap.
|
||||
- // On the MUL_MAT_ID grouped-GEMM path (expert_bounds != nullptr) the GEMM columns are
|
||||
- // tokens sorted by expert; stock picks mmq_x to cover ncols_max (= ne12, the token count,
|
||||
- // up to 128) in a single column-tile. At MoE decode the per-expert token density is low
|
||||
- // (top-k of many experts: ~ne12*k/n_experts tokens/expert, e.g. ~8 at npl128 for
|
||||
- // Qwen3-30B-A3B top-8/128), so each expert's single mmq_x-wide col-tile is mostly empty:
|
||||
- // the MMA accumulator tile is mmq_x-wide at compile time and wastes throughput on the
|
||||
- // padding columns while the larger y-tile lowers occupancy. Capping mmq_x toward the
|
||||
- // per-expert density raises tile fill + occupancy with no extra weight reads (at
|
||||
- // tokens/expert <= mmq_x there is still exactly one non-empty col-tile per expert; the
|
||||
- // emptier tiles are skipped by the jt*mmq_x >= col_diff guard in the stream-k kernel).
|
||||
- // Default (env unset or <= 0) = disabled => mmq_x selection is byte-identical to stock;
|
||||
- // off the ids path the cap never applies.
|
||||
+ // [paged patch 0015] expert-density-aware MoE token-tile (mmq_x) auto-select (DEFAULT-ON).
|
||||
+ // On the MUL_MAT_ID grouped-GEMM path (expert_bounds != nullptr) the GEMM columns are tokens
|
||||
+ // sorted by expert; stock picks mmq_x to cover ncols_max (= ne12, the token count, up to 128)
|
||||
+ // in a single column-tile, i.e. it MAXIMIZES the tile (128 on Blackwell) for the aggregate
|
||||
+ // batch. But the tile is then applied PER EXPERT, and at MoE decode the per-expert token
|
||||
+ // density is tiny (top-k of many experts), so each expert's single 128-wide col-tile is mostly
|
||||
+ // empty: the MMA accumulator tile is mmq_x-wide at compile time and burns throughput on the
|
||||
+ // padding columns while the larger y-tile lowers occupancy. vLLM's fused-MoE does the opposite
|
||||
+ // (a small per-expert BLOCK_SIZE_M). We reproduce that here, host-side only, by picking a
|
||||
+ // SMALLER mmq_x when - and only when - the per-expert density is low:
|
||||
+ //
|
||||
+ // ne_get_rows = args.ncols_dst = ne12 * n_expert_used (total token-expert assignments)
|
||||
+ // n_experts = args.nchannels_x = ne02
|
||||
+ // n_active_est = min(n_experts, ne_get_rows) (upper bound on active experts)
|
||||
+ // density = ceil(ne_get_rows / n_active_est) (avg tokens per active expert)
|
||||
+ //
|
||||
+ // Cap to the small tile (default 64) only when density <= density_max (default 8). 8 sits below
|
||||
+ // every prefill-ubatch density and above every decode density for n_experts in [128,511] at the
|
||||
+ // standard n_ubatch=512 (prefill 4096/n_experts, decode <=1024/n_experts), with >=8x fill headroom
|
||||
+ // so a capped expert segment never splits a col-tile. Decode (per-expert density 4 at 256 experts,
|
||||
+ // 8 at 128 experts @npl128) gets the fuller high-occupancy tile; the prefill ubatch (density 16 at
|
||||
+ // 256 / 32 at 128 experts) stays ABOVE the threshold and keeps the big
|
||||
+ // 128 compute tile - so unlike the blunt global cap (LLAMA_MOE_MMQ_X / patch 0014) this is
|
||||
+ // prefill-safe by construction. The selection only ever picks an already-compiled, granularity-
|
||||
+ // and shared-memory-validated mmq_x that the loop below would consider for a smaller batch; no
|
||||
+ // new kernel. Off the ids path (expert_bounds == nullptr) nothing changes => non-MoE mul_mat
|
||||
+ // and the gated f16/bf16 host-loop fallback stay byte-identical to stock.
|
||||
+ // - LLAMA_MOE_MMQ_X=<n> : manual blunt global cap, overrides the auto-select (patch 0014).
|
||||
+ // - LLAMA_MOE_AUTO_TILE=0 : disable the auto-select (exact stock selection).
|
||||
+ // - LLAMA_MOE_DECODE_TILE=<n>, LLAMA_MOE_DENSITY_MAX=<n> : tune the tile / threshold.
|
||||
int mmq_x_lim = mmq_x_max;
|
||||
if (args.expert_bounds != nullptr) {
|
||||
const int moe_cap = ggml_cuda_moe_mmq_x_cap();
|
||||
if (moe_cap > 0) {
|
||||
const int cap = moe_cap < 8 ? 8 : moe_cap;
|
||||
mmq_x_lim = cap < mmq_x_max ? cap : mmq_x_max;
|
||||
+ } else if (ggml_cuda_moe_auto_tile_enabled()) {
|
||||
+ const int64_t ne_get_rows = args.ncols_dst;
|
||||
+ const int64_t n_experts = args.nchannels_x;
|
||||
+ if (ne_get_rows > 0 && n_experts > 0) {
|
||||
+ const int64_t n_active = ne_get_rows < n_experts ? ne_get_rows : n_experts;
|
||||
+ const int64_t density = (ne_get_rows + n_active - 1) / n_active;
|
||||
+ const int tile = ggml_cuda_moe_decode_tile();
|
||||
+ if (density <= (int64_t) ggml_cuda_moe_density_max() && tile < mmq_x_max) {
|
||||
+ mmq_x_lim = tile;
|
||||
+ }
|
||||
+ }
|
||||
}
|
||||
}
|
||||
|
||||
diff --git a/tests/test-backend-ops.cpp b/tests/test-backend-ops.cpp
|
||||
index c83e91f..62a0989 100644
|
||||
--- a/tests/test-backend-ops.cpp
|
||||
+++ b/tests/test-backend-ops.cpp
|
||||
@@ -8603,6 +8603,22 @@ static std::vector<std::unique_ptr<test_case>> make_test_cases_eval() {
|
||||
test_cases.emplace_back(new test_mul_mat_id(GGML_TYPE_MXFP4, GGML_TYPE_F32, 32, 2, false, 2880, 32, 2880));
|
||||
test_cases.emplace_back(new test_mul_mat_id(GGML_TYPE_Q4_0, GGML_TYPE_F32, 32, 2, false, 2880, 32, 2880));
|
||||
|
||||
+ // [paged P0] MXFP4/NVFP4 qwen3-30b-a3b MoE decode-density regression gate for the expert-
|
||||
+ // density-aware mmq_x auto-select (patch 0015). Real expert-FFN slice (128 experts, top-8,
|
||||
+ // m=768, k=2048) so this exercises the exact grouped FP4-MMA mmq kernel the model runs.
|
||||
+ // Per-expert token density = n*n_used/n_mats = n/16; cover the decode band (density 1/4/8/16
|
||||
+ // at n 16/64/128/256), ragged token counts (n 33/130/200: experts with 0/1/2 tokens, n not a
|
||||
+ // multiple of the tile) where the tiny-M col-tiles change geometry and any masking can leak,
|
||||
+ // and a prefill-density shape (n 512 => density 32) the auto-select must leave on the large
|
||||
+ // 128 tile. n>=128 is exactly where stock picks mmq_x=128 and the auto-select picks 64, so the
|
||||
+ // op-test (CPU oracle vs CUDA, deterministic) is the bit-exact regression gate for P1: it must
|
||||
+ // pass with the auto-select on (default) and with LLAMA_MOE_AUTO_TILE=0 (stock selection).
|
||||
+ for (ggml_type type_a : {GGML_TYPE_MXFP4, GGML_TYPE_NVFP4}) {
|
||||
+ for (int n : {16, 33, 64, 128, 130, 200, 256, 512}) {
|
||||
+ test_cases.emplace_back(new test_mul_mat_id(type_a, GGML_TYPE_F32, 128, 8, false, 768, n, 2048));
|
||||
+ }
|
||||
+ }
|
||||
+
|
||||
for (ggml_type type_a : all_types) {
|
||||
test_cases.emplace_back(new test_mul_mat_id(type_a, GGML_TYPE_F32, 4, 2, false, 64, 16, 3*ggml_blck_size(type_a)));
|
||||
}
|
||||
--
|
||||
2.43.0
|
||||
|
||||
@@ -1,191 +0,0 @@
|
||||
From 02fa0473a9324b7e12f9b203d221cc4ac80cfd33 Mon Sep 17 00:00:00 2001
|
||||
From: Ettore Di Giacinto <mudler@localai.io>
|
||||
Date: Wed, 24 Jun 2026 10:11:48 +0200
|
||||
Subject: [PATCH] feat(paged): dynamic decode-first prefill-token budget (patch
|
||||
0016, continuous-batch P1)
|
||||
|
||||
Supersede patch 0013's STATIC per-step prefill cap with a DYNAMIC,
|
||||
decode-first token budget: the P1 of the token-granular continuous-batch
|
||||
scheduler. POLICY change only inside update_slots(): no new slot states, no
|
||||
batch-formation rewrite, zero libllama changes. llama-server already emits one
|
||||
unified mixed prefill+decode batch per step (Phase 1 appends every ready decode
|
||||
token unconditionally; Phase 2 fills prefill into the same batch). 0016 only
|
||||
changes the COUNT of prefill tokens admitted per step.
|
||||
|
||||
The budget block already sits AFTER Phase 1's decode fill, so batch.n_tokens
|
||||
== D (the live decode load) is known there. Instead of 0013's constant
|
||||
LLAMA_PREFILL_BUDGET (which ignores D, needs per-workload tuning, and lets one
|
||||
long prompt monopolise the step), compute a dynamic budget:
|
||||
|
||||
T = clamp(LLAMA_MAX_BATCH_TOKENS (default n_batch), n_ubatch, n_batch)
|
||||
prefill_budget_step = max(n_ubatch, T - D) (leftover after decode,
|
||||
auto-shrinks as decode load rises so the step never inflates past T)
|
||||
prefill_cap_per_slot = min(T, ceil(0.04*n_ctx)) floored at n_ubatch,
|
||||
pinned to n_batch when T == n_batch (LLAMA_PREFILL_CAP overrides)
|
||||
|
||||
Phase 2's inner prompt-fill loop and outer admission break are bounded by
|
||||
prefill_budget_step (across slots) and a new per-slot slot_prompt_added
|
||||
counter; the n_batch hard ceiling stays as the compute bound. Decode is
|
||||
structurally claimed first and never capped (Phase 1), so the decode-first
|
||||
guarantee is free.
|
||||
|
||||
DEFAULT-OFF BYTE-IDENTICAL: with all knobs unset, behaviour is byte-identical
|
||||
to stock. The degenerate T == n_batch case is byte-identical to stock/0013 (the
|
||||
determinism oracle). The legacy LLAMA_PREFILL_BUDGET path is preserved exactly
|
||||
(honoured only when LLAMA_MAX_BATCH_TOKENS is unset), so 0013 is cleanly
|
||||
subsumed. Orthogonal to LLAMA_KV_PAGED: pure scheduler policy, identical
|
||||
decisions paged on or off.
|
||||
|
||||
Assisted-by: Claude:opus-4.8 [Claude Code]
|
||||
Signed-off-by: Ettore Di Giacinto <mudler@localai.io>
|
||||
---
|
||||
tools/server/server-context.cpp | 107 +++++++++++++++++++++++++-------
|
||||
1 file changed, 85 insertions(+), 22 deletions(-)
|
||||
|
||||
diff --git a/tools/server/server-context.cpp b/tools/server/server-context.cpp
|
||||
index afcdebe..b8b8f00 100644
|
||||
--- a/tools/server/server-context.cpp
|
||||
+++ b/tools/server/server-context.cpp
|
||||
@@ -3043,24 +3043,78 @@ private:
|
||||
int32_t n_batch = llama_n_batch(ctx_tgt);
|
||||
int32_t n_ubatch = llama_n_ubatch(ctx_tgt);
|
||||
|
||||
- // PAGED serving lever (patch 0013): decoupled per-step prefill-token budget.
|
||||
- // Analogue of vLLM's --max-num-batched-tokens. Stock llama-server caps the prompt
|
||||
- // tokens ingested per update_slots() step at n_batch only; with cont_batching the
|
||||
- // sampled decode tokens of every generating slot are appended FIRST, then prompt
|
||||
- // tokens fill the batch up to n_batch. A long prompt therefore grabs an ~n_batch
|
||||
- // chunk in a SINGLE compute-heavy step, spiking the inter-token latency of every
|
||||
- // co-batched decoder (head-of-line jitter). LLAMA_PREFILL_BUDGET caps the prompt
|
||||
- // tokens added per step independently of n_batch, splitting a long prefill across
|
||||
- // more steps so in-flight decode keeps advancing smoothly. Default (env unset or
|
||||
- // <=0) = disabled => stock behavior is byte-identical. Orthogonal to LLAMA_KV_PAGED
|
||||
- // (this is a pure scheduler knob; works with paged off).
|
||||
- int32_t n_prefill_budget = 0; // 0 = disabled (stock n_batch-only chunking)
|
||||
+ // PAGED serving lever (patch 0016, supersedes 0013): dynamic decode-first
|
||||
+ // per-step prefill-token budget (continuous-batch scheduler P1). llama-server
|
||||
+ // already builds ONE mixed batch per update_slots() step: Phase 1 (just above)
|
||||
+ // appended every generating slot's sampled token UNCONDITIONALLY, so at this point
|
||||
+ // batch.n_tokens == D is the live decode load; Phase 2 (below) fills the remaining
|
||||
+ // batch capacity with prompt tokens. Patch 0013 capped Phase 2 with a STATIC
|
||||
+ // constant (LLAMA_PREFILL_BUDGET) that ignores D, needs per-workload tuning, and
|
||||
+ // lets one long prompt monopolise the step.
|
||||
+ //
|
||||
+ // This computes a DYNAMIC budget instead, the vLLM v1 token-budget analogue:
|
||||
+ // a single total per-step token budget T, decode claims its D tokens first
|
||||
+ // (already in the batch), and prefill gets the leftover T - D distributed across
|
||||
+ // waiting prompts with a per-slot chunk cap. As decode load D rises the prefill
|
||||
+ // leftover auto-shrinks, so the step never inflates past T at any concurrency:
|
||||
+ // the budget self-tunes across the npl range and across dense vs MoE without a
|
||||
+ // hand-picked constant (the 161/333 tok/s GB10 decode ceiling is held tuning-free
|
||||
+ // instead of via 0013's hand-tuned 256). Decode is structurally claimed first and
|
||||
+ // never capped (Phase 1), so the decode-first guarantee is free here.
|
||||
+ //
|
||||
+ // LLAMA_MAX_BATCH_TOKENS (T) total per-step token budget (decode + prefill),
|
||||
+ // default n_batch, clamped to [n_ubatch, n_batch] so
|
||||
+ // the compute loop stays a single llama_decode and
|
||||
+ // prefill keeps an n_ubatch floor of progress.
|
||||
+ // LLAMA_PREFILL_CAP per-slot max prompt tokens per step (the
|
||||
+ // long_prefill_token_threshold analogue), default
|
||||
+ // min(T, ceil(0.04*n_ctx)) floored at n_ubatch, so
|
||||
+ // one long prompt cannot eat the whole leftover.
|
||||
+ // LLAMA_PREFILL_BUDGET legacy static cap (patch 0013); honoured ONLY when
|
||||
+ // LLAMA_MAX_BATCH_TOKENS is unset, for back-compat.
|
||||
+ //
|
||||
+ // DEFAULT-OFF BYTE-IDENTICAL: with all three knobs unset, and in the degenerate
|
||||
+ // T == n_batch case, behaviour is byte-identical to stock. At T == n_batch the
|
||||
+ // dynamic leftover max(n_ubatch, n_batch - D) and the n_batch per-slot cap both
|
||||
+ // reach the existing `batch.n_tokens < n_batch` ceiling at the SAME point, so no
|
||||
+ // new bound fires (the determinism oracle). Orthogonal to LLAMA_KV_PAGED: pure
|
||||
+ // scheduler policy, identical decisions with paged on or off.
|
||||
+ const int32_t n_decode_in_batch = batch.size(); // D: Phase 1 appended D decode tokens above
|
||||
+ int32_t prefill_budget_step = 0; // 0 = disabled (stock n_batch-only chunking)
|
||||
+ int32_t prefill_cap_per_slot = 0; // 0 = disabled (no per-slot prompt-chunk cap)
|
||||
{
|
||||
- const char * env_pb = getenv("LLAMA_PREFILL_BUDGET");
|
||||
- if (env_pb) {
|
||||
+ int32_t mbt = 0;
|
||||
+ if (const char * env_mbt = getenv("LLAMA_MAX_BATCH_TOKENS")) {
|
||||
+ mbt = atoi(env_mbt);
|
||||
+ }
|
||||
+ if (mbt > 0) {
|
||||
+ // dynamic decode-first budget (P1): T clamped to [n_ubatch, n_batch]
|
||||
+ int32_t T = std::min(n_batch, mbt);
|
||||
+ T = std::max(T, n_ubatch);
|
||||
+ // leftover after decode, floored at n_ubatch so prefill never fully starves
|
||||
+ prefill_budget_step = std::max(n_ubatch, T - n_decode_in_batch);
|
||||
+ // per-slot prompt-chunk cap (long_prefill_token_threshold analogue)
|
||||
+ int32_t cap = 0;
|
||||
+ if (const char * env_cap = getenv("LLAMA_PREFILL_CAP")) {
|
||||
+ cap = atoi(env_cap);
|
||||
+ }
|
||||
+ if (cap <= 0) {
|
||||
+ const int32_t pct4 = (n_ctx + 24) / 25; // ceil(0.04 * n_ctx)
|
||||
+ cap = std::min(T, std::max(n_ubatch, pct4));
|
||||
+ }
|
||||
+ cap = std::min(n_batch, std::max(n_ubatch, cap));
|
||||
+ // at T == n_batch the leftover and cap both reach the n_batch ceiling
|
||||
+ // together; pin the cap to n_batch so this case stays byte-identical
|
||||
+ if (T >= n_batch) {
|
||||
+ cap = n_batch;
|
||||
+ }
|
||||
+ prefill_cap_per_slot = cap;
|
||||
+ } else if (const char * env_pb = getenv("LLAMA_PREFILL_BUDGET")) {
|
||||
+ // legacy static budget (patch 0013), kept for back-compat when the
|
||||
+ // dynamic knob is unset: a constant per-step prefill cap, no per-slot cap
|
||||
const int v = atoi(env_pb);
|
||||
if (v > 0) {
|
||||
- n_prefill_budget = std::min(n_batch, std::max(1, v));
|
||||
+ prefill_budget_step = std::min(n_batch, std::max(1, v));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -3509,11 +3563,18 @@ private:
|
||||
const auto & spans = slot.task->params.message_spans;
|
||||
const auto last_user_pos = spans.last_user_message_pos();
|
||||
|
||||
+ // (patch 0016) per-slot prompt tokens added this step, for the per-slot
|
||||
+ // chunk cap (resets each slot); n_batch stays the hard compute ceiling
|
||||
+ int32_t slot_prompt_added = 0;
|
||||
+
|
||||
// add prompt tokens for processing in the current batch
|
||||
- // (patch 0013) also stop once the per-step prefill budget is spent, so a long
|
||||
- // prompt is split across more steps and leaves batch room for co-batched decode
|
||||
+ // (patch 0016) also stop once (a) the dynamic per-step prefill budget
|
||||
+ // (the T - D leftover) is spent across all slots, or (b) this slot's
|
||||
+ // per-slot chunk cap is hit, so a long prompt is split across more steps
|
||||
+ // and leaves batch room for co-batched decode of the other slots
|
||||
while (slot.prompt.n_tokens() < slot.task->n_tokens() && batch.size() < n_batch &&
|
||||
- (n_prefill_budget == 0 || n_prompt_budgeted < n_prefill_budget)) {
|
||||
+ (prefill_budget_step == 0 || n_prompt_budgeted < prefill_budget_step) &&
|
||||
+ (prefill_cap_per_slot == 0 || slot_prompt_added < prefill_cap_per_slot)) {
|
||||
// get next token to process
|
||||
llama_token cur_tok = input_tokens[slot.prompt.n_tokens()];
|
||||
if (cur_tok == LLAMA_TOKEN_NULL) {
|
||||
@@ -3538,7 +3599,8 @@ private:
|
||||
slot.prompt.tokens.push_back(cur_tok);
|
||||
|
||||
slot.n_prompt_tokens_processed++;
|
||||
- n_prompt_budgeted++; // (patch 0013) count toward the per-step prefill budget
|
||||
+ n_prompt_budgeted++; // (patch 0016) toward the dynamic per-step prefill budget
|
||||
+ slot_prompt_added++; // (patch 0016) toward this slot's per-step chunk cap
|
||||
|
||||
// stop the prompt batch exactly before a user message
|
||||
if (spans.is_user_start(slot.prompt.n_tokens())) {
|
||||
@@ -3624,9 +3686,10 @@ private:
|
||||
if (!slot_batched) {
|
||||
slot_batched = &slot;
|
||||
}
|
||||
- // (patch 0013) stop adding prompts once the per-step prefill budget is spent,
|
||||
- // leaving the remaining batch capacity for co-batched decode of other slots
|
||||
- if (n_prefill_budget > 0 && n_prompt_budgeted >= n_prefill_budget) {
|
||||
+ // (patch 0016) stop admitting prompts once the dynamic per-step prefill
|
||||
+ // budget (the T - D leftover) is spent, leaving the remaining batch
|
||||
+ // capacity for co-batched decode of the other slots
|
||||
+ if (prefill_budget_step > 0 && n_prompt_budgeted >= prefill_budget_step) {
|
||||
add_ok = false;
|
||||
}
|
||||
});
|
||||
--
|
||||
2.43.0
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user