mirror of
https://github.com/mudler/LocalAI.git
synced 2026-07-31 10:28:43 -04:00
* feat(backends): add LongCat video and avatar generation Assisted-by: Codex:GPT-5 [apply_patch] [exec_command] [web] * refactor(config): declare model I/O modalities Make model configs declare input and output modalities so capability discovery no longer branches on backend or checkpoint names. Complete the LongCat gallery and user documentation, make the SDPA patch apply to the pinned upstream revision, and stabilize the Agent Jobs race exposed by the required hook. Assisted-by: Codex:GPT-5 [web] --------- Co-authored-by: Ettore Di Giacinto <mudler@localai.io>
76 lines
3.0 KiB
Diff
76 lines
3.0 KiB
Diff
diff --git a/longcat_video/modules/attention.py b/longcat_video/modules/attention.py
|
|
index bb5630f..9b9f3cc 100644
|
|
--- a/longcat_video/modules/attention.py
|
|
+++ b/longcat_video/modules/attention.py
|
|
@@ -2,6 +2,7 @@ from typing import List, Optional
|
|
|
|
import torch
|
|
import torch.nn as nn
|
|
+import torch.nn.functional as F
|
|
|
|
from einops import rearrange
|
|
|
|
@@ -100,7 +101,8 @@ class Attention(nn.Module):
|
|
x = xformers.ops.memory_efficient_attention(q, k, v, attn_bias=None, op=None,)
|
|
x = rearrange(x, "B M H K -> B H M K")
|
|
else:
|
|
- raise RuntimeError("Unsupported attention operations.")
|
|
+ # Keep a dependency-free path for systems without optional kernels.
|
|
+ x = F.scaled_dot_product_attention(q, k, v, scale=self.scale)
|
|
|
|
return x
|
|
|
|
@@ -245,8 +247,22 @@ class MultiHeadCrossAttention(nn.Module):
|
|
attn_bias = xformers.ops.fmha.attn_bias.BlockDiagonalMask.from_seqlens([N] * B, kv_seqlen)
|
|
x = xformers.ops.memory_efficient_attention(q, k, v, attn_bias=attn_bias)
|
|
else:
|
|
- raise RuntimeError("Unsupported attention operations.")
|
|
-
|
|
+ # Preserve the variable-length block boundaries without materializing
|
|
+ # a dense attention mask.
|
|
+ blocks = []
|
|
+ offset = 0
|
|
+ for batch_index, key_count in enumerate(kv_seqlen):
|
|
+ query = q[0][batch_index * N:(batch_index + 1) * N]
|
|
+ key = k[0][offset:offset + key_count]
|
|
+ value = v[0][offset:offset + key_count]
|
|
+ output = F.scaled_dot_product_attention(
|
|
+ query.transpose(0, 1),
|
|
+ key.transpose(0, 1),
|
|
+ value.transpose(0, 1),
|
|
+ )
|
|
+ blocks.append(output.transpose(0, 1))
|
|
+ offset += key_count
|
|
+ x = torch.cat(blocks, dim=0)
|
|
|
|
x = x.view(B, -1, C)
|
|
x = self.proj(x)
|
|
diff --git a/longcat_video/modules/avatar/attention.py b/longcat_video/modules/avatar/attention.py
|
|
index a169a7a..df9a469 100644
|
|
--- a/longcat_video/modules/avatar/attention.py
|
|
+++ b/longcat_video/modules/avatar/attention.py
|
|
@@ -2,6 +2,7 @@ from typing import List, Optional
|
|
|
|
import torch
|
|
import torch.nn as nn
|
|
+import torch.nn.functional as F
|
|
|
|
from einops import rearrange
|
|
|
|
@@ -111,7 +112,8 @@ class Attention(nn.Module):
|
|
x = xformers.ops.memory_efficient_attention(q, k, v, attn_bias=None, op=None,)
|
|
x = rearrange(x, "B M H K -> B H M K")
|
|
else:
|
|
- raise RuntimeError("Unsupported attention operations.")
|
|
+ # Keep a dependency-free path for systems without optional kernels.
|
|
+ x = F.scaled_dot_product_attention(q, k, v, scale=self.scale)
|
|
|
|
return x
|
|
|
|
@@ -429,2 +431,5 @@ class SingleStreamAttention(nn.Module):
|
|
+ else:
|
|
+ # This branch uses the native PyTorch kernel when optional kernels are off.
|
|
+ x = F.scaled_dot_product_attention(q, encoder_k, encoder_v, scale=self.scale)
|
|
|
|
# linear transform
|