mirror of
https://github.com/mudler/LocalAI.git
synced 2026-09-21 05:34:56 -04:00
* fix(vulkan): preserve host ICD discovery for packaged backends Add bundled Mesa manifests through VK_ADD_DRIVER_FILES instead of replacing the system driver list. Merge inherited and model-specific additive paths while preserving explicit operator overrides, with regression coverage. Assisted-by: Codex:gpt-5 golangci-lint Assisted-by: Codex:gpt-5.6-sol Signed-off-by: Richard Palethorpe <io@richiejp.com> * feat(3d): add Kimodo CPU and Vulkan animation backend Introduce a distinct animation capability and model-described 3D operations, with a typed /3d/animate API, RPC transport, distributed media staging, permissions, and tracing. Add a persistent kimodo.cpp adapter, skeleton GLB export, CPU/Vulkan packages, model and backend galleries, importer support, CI builds, and documentation. Adapt Studio inputs to each model and provide real-time skeleton playback, seeking, and history. Cover backend validation, packaging, API behavior, importer inventories, distributed staging, and Studio workflows. Validate real-model CPU/Vulkan generation and deploy the integration to the local QA instance. Assisted-by: Codex:gpt-5 golangci-lint Assisted-by: Codex:gpt-5.6-sol Signed-off-by: Richard Palethorpe <io@richiejp.com> * feat(kimodocpp): adopt monolithic encoders and resident inference Update upstream for resident weights, packed execution paths, and cached motion graphs. Default to all 32 text layers while retaining configurable streaming and legacy bundle support. Use monolithic Q8_0 encoders by default and offer all six published quantizations through the gallery and importer. Refresh pinned hashes, tests, and documentation; remove the obsolete thread patch and ensure cached source checkouts follow the upstream pin. Validated CPU and Vulkan generation, lower-bit streaming, gallery/importer suites, packaging, lint, and cold/warm Studio generation on localai-dev. Assisted-by: Codex:gpt-5 golangci-lint Assisted-by: Codex:gpt-5.6-sol Signed-off-by: Richard Palethorpe <io@richiejp.com> --------- Signed-off-by: Richard Palethorpe <io@richiejp.com>
100 lines
4.3 KiB
Go
100 lines
4.3 KiB
Go
package grpc
|
|
|
|
import (
|
|
"context"
|
|
|
|
pb "github.com/mudler/LocalAI/pkg/grpc/proto"
|
|
)
|
|
|
|
type AIModel interface {
|
|
Busy() bool
|
|
Lock()
|
|
Unlock()
|
|
Locking() bool
|
|
Predict(*pb.PredictOptions) (string, error)
|
|
PredictStream(*pb.PredictOptions, chan string) error
|
|
Load(*pb.ModelOptions) error
|
|
Free() error
|
|
Embeddings(*pb.PredictOptions) ([]float32, error)
|
|
GenerateImage(*pb.GenerateImageRequest) error
|
|
UpscaleImage(*pb.UpscaleImageRequest) error
|
|
GenerateVideo(*pb.GenerateVideoRequest) error
|
|
Generate3D(*pb.Generate3DRequest) error
|
|
Animate3D(*pb.Animate3DRequest) error
|
|
Detect(*pb.DetectOptions) (pb.DetectResponse, error)
|
|
Depth(*pb.DepthRequest) (pb.DepthResponse, error)
|
|
FaceVerify(*pb.FaceVerifyRequest) (pb.FaceVerifyResponse, error)
|
|
FaceAnalyze(*pb.FaceAnalyzeRequest) (pb.FaceAnalyzeResponse, error)
|
|
VoiceVerify(*pb.VoiceVerifyRequest) (pb.VoiceVerifyResponse, error)
|
|
VoiceAnalyze(*pb.VoiceAnalyzeRequest) (pb.VoiceAnalyzeResponse, error)
|
|
VoiceEmbed(*pb.VoiceEmbedRequest) (pb.VoiceEmbedResponse, error)
|
|
AudioTranscription(context.Context, *pb.TranscriptRequest) (pb.TranscriptResult, error)
|
|
AudioTranscriptionStream(context.Context, *pb.TranscriptRequest, chan *pb.TranscriptStreamResponse) error
|
|
AudioTranscriptionLive(in <-chan *pb.TranscriptLiveRequest, out chan<- *pb.TranscriptLiveResponse) error
|
|
TTS(*pb.TTSRequest) error
|
|
TTSStream(*pb.TTSRequest, chan []byte) error
|
|
SoundGeneration(*pb.SoundGenerationRequest) error
|
|
TokenizeString(*pb.PredictOptions) (pb.TokenizationResponse, error)
|
|
Detokenize(*pb.DetokenizeRequest) (pb.DetokenizeResponse, error)
|
|
Status() (pb.StatusResponse, error)
|
|
|
|
StoresSet(*pb.StoresSetOptions) error
|
|
StoresDelete(*pb.StoresDeleteOptions) error
|
|
StoresGet(*pb.StoresGetOptions) (pb.StoresGetResult, error)
|
|
StoresFind(*pb.StoresFindOptions) (pb.StoresFindResult, error)
|
|
|
|
VAD(*pb.VADRequest) (pb.VADResponse, error)
|
|
Diarize(*pb.DiarizeRequest) (pb.DiarizeResponse, error)
|
|
SoundDetection(context.Context, *pb.SoundDetectionRequest) (*pb.SoundDetectionResponse, error)
|
|
|
|
AudioEncode(*pb.AudioEncodeRequest) (*pb.AudioEncodeResult, error)
|
|
AudioDecode(*pb.AudioDecodeRequest) (*pb.AudioDecodeResult, error)
|
|
|
|
AudioTransform(*pb.AudioTransformRequest) (*pb.AudioTransformResult, error)
|
|
AudioTransformStream(in <-chan *pb.AudioTransformFrameRequest, out chan<- *pb.AudioTransformFrameResponse) error
|
|
AudioToAudioStream(in <-chan *pb.AudioToAudioRequest, out chan<- *pb.AudioToAudioResponse) error
|
|
|
|
// Forward proxies a raw HTTP request to an upstream provider for
|
|
// passthrough-mode cloud-proxy backends. ctx is the gRPC stream
|
|
// context — cancellation propagates to the upstream HTTP request
|
|
// so client disconnect closes the upstream connection.
|
|
Forward(ctx context.Context, in <-chan *pb.ForwardRequest, out chan<- *pb.ForwardReply) error
|
|
|
|
ModelMetadata(*pb.ModelOptions) (*pb.ModelMetadataResponse, error)
|
|
|
|
// Fine-tuning
|
|
StartFineTune(*pb.FineTuneRequest) (*pb.FineTuneJobResult, error)
|
|
FineTuneProgress(*pb.FineTuneProgressRequest, chan *pb.FineTuneProgressUpdate) error
|
|
StopFineTune(*pb.FineTuneStopRequest) error
|
|
ListCheckpoints(*pb.ListCheckpointsRequest) (*pb.ListCheckpointsResponse, error)
|
|
ExportModel(*pb.ExportModelRequest) error
|
|
|
|
// Quantization
|
|
StartQuantization(*pb.QuantizationRequest) (*pb.QuantizationJobResult, error)
|
|
QuantizationProgress(*pb.QuantizationProgressRequest, chan *pb.QuantizationProgressUpdate) error
|
|
StopQuantization(*pb.QuantizationStopRequest) error
|
|
}
|
|
|
|
func newReply(s string) *pb.Reply {
|
|
return &pb.Reply{Message: []byte(s)}
|
|
}
|
|
|
|
// AIModelRich is an optional extension to AIModel for backends that
|
|
// can produce a full *pb.Reply — including tool-call deltas and
|
|
// usage tokens — rather than just a content string. The gRPC server
|
|
// type-asserts and prefers the rich path when implemented; otherwise
|
|
// it wraps Predict's string return in a Reply.
|
|
//
|
|
// Cloud-proxy translate mode is the motivating use case: the upstream
|
|
// emits structured tool_calls that would be lost through the legacy
|
|
// (string, error) signature.
|
|
//
|
|
// PredictStreamRich contract: send replies into the channel and
|
|
// return when finished. Do NOT close the channel — the server closes
|
|
// it after the call returns. This is opposite to legacy PredictStream
|
|
// which expects the impl to defer close().
|
|
type AIModelRich interface {
|
|
PredictRich(*pb.PredictOptions) (*pb.Reply, error)
|
|
PredictStreamRich(*pb.PredictOptions, chan<- *pb.Reply) error
|
|
}
|