Enhancements (#34 )

Signed-off-by: mudler <mudler@c3os.io>
⬆️ Bump llama.cpp (#33 )
2026-02-04 03:32:40 -05:00 · 2023-04-19 17:10:29 +02:00 · 2023-04-17 21:34:02 +02:00 · 2023-04-17 18:45:42 +02:00 · 2023-04-16 10:46:20 +02:00 · 2023-04-16 10:40:50 +02:00
20 changed files with 379 additions and 469 deletions
--- a/.dockerignore
+++ b/.dockerignore
@@ -1 +1 @@
-models/*.bin
+models
--- a/.env
+++ b/.env
@@ -1 +1,3 @@
-THREADS=14
+THREADS=14
+CONTEXT_SIZE=512
+MODELS_PATH=/models
--- a/.gitignore
+++ b/.gitignore
@@ -1,2 +1,10 @@
+# go-llama build artifacts
+go-llama
+go-gpt4all-j
+
+# llama-cli build binary
 llama-cli
-models/*.bin
+
+# Ignore models
+models/*.bin
+models/ggml-*
--- a/.vscode/launch.json
+++ b/.vscode/launch.json
@@ -0,0 +1,16 @@
+{
+    "version": "0.2.0",
+    "configurations": [
+    
+    {
+        "name": "Launch Go",
+        "type": "go",
+        "request": "launch",
+        "mode": "debug",
+        "program": "${workspaceFolder}/main.go",
+        "args": [
+            "api"
+        ]
+    }
+    ]
+}
--- a/12
+++ b/12
@@ -2,16 +2,10 @@ ARG GO_VERSION=1.20
 ARG DEBIAN_VERSION=11
 FROM golang:$GO_VERSION as builder
 WORKDIR /build
-ARG GO_LLAMA_CPP_TAG=llama.cpp-2f7c8e0
-RUN git clone -b $GO_LLAMA_CPP_TAG --recurse-submodules https://github.com/go-skynet/go-llama.cpp
-RUN cd go-llama.cpp && make libbinding.a
-COPY go.mod ./
-COPY go.sum ./
-RUN go mod download
-RUN apt-get update
+RUN apt-get update && apt-get install -y cmake
 COPY . .
-RUN go mod edit -replace github.com/go-skynet/go-llama.cpp=/build/go-llama.cpp
-RUN C_INCLUDE_PATH=/build/go-llama.cpp LIBRARY_PATH=/build/go-llama.cpp go build -o llama-cli ./
+ARG BUILD_TYPE=
+RUN make build${BUILD_TYPE}

 FROM debian:$DEBIAN_VERSION
 COPY --from=builder /build/llama-cli /usr/bin/llama-cli
--- a/79
+++ b/79
@@ -0,0 +1,79 @@
+GOCMD=go
+GOTEST=$(GOCMD) test
+GOVET=$(GOCMD) vet
+BINARY_NAME=llama-cli
+GOLLAMA_VERSION?=llama.cpp-5ecff35
+
+GREEN  := $(shell tput -Txterm setaf 2)
+YELLOW := $(shell tput -Txterm setaf 3)
+WHITE  := $(shell tput -Txterm setaf 7)
+CYAN   := $(shell tput -Txterm setaf 6)
+RESET  := $(shell tput -Txterm sgr0)
+
+.PHONY: all test build vendor
+
+all: help
+
+## Build:
+
+build: prepare ## Build the project
+	C_INCLUDE_PATH=$(shell pwd)/go-llama.cpp:$(shell pwd)/go-gpt4all-j LIBRARY_PATH=$(shell pwd)/go-llama.cpp:$(shell pwd)/go-gpt4all-j $(GOCMD) build -o $(BINARY_NAME) ./
+
+buildgeneric: prepare-generic ## Build the project
+	C_INCLUDE_PATH=$(shell pwd)/go-llama.cpp:$(shell pwd)/go-gpt4all-j LIBRARY_PATH=$(shell pwd)/go-llama.cpp:$(shell pwd)/go-gpt4all-j $(GOCMD) build -o $(BINARY_NAME) ./
+
+go-gpt4all-j:
+	git clone --recurse-submodules https://github.com/go-skynet/go-gpt4all-j.cpp go-gpt4all-j
+# This is hackish, but needed as both go-llama and go-gpt4allj have their own version of ggml..
+	@find ./go-gpt4all-j -type f -name "*.c" -exec sed -i'' -e 's/ggml_/ggml_gptj_/g' {} +
+	@find ./go-gpt4all-j -type f -name "*.cpp" -exec sed -i'' -e 's/ggml_/ggml_gptj_/g' {} +
+	@find ./go-gpt4all-j -type f -name "*.h" -exec sed -i'' -e 's/ggml_/ggml_gptj_/g' {} +
+	@find ./go-gpt4all-j -type f -name "*.cpp" -exec sed -i'' -e 's/gpt_/gptj_/g' {} +
+	@find ./go-gpt4all-j -type f -name "*.h" -exec sed -i'' -e 's/gpt_/gptj_/g' {} +
+
+go-gpt4all-j/libgptj.a: go-gpt4all-j
+	$(MAKE) -C go-gpt4all-j libgptj.a
+
+go-gpt4all-j/libgptj.a-generic: go-gpt4all-j
+	$(MAKE) -C go-gpt4all-j generic-libgptj.a
+
+go-llama:
+	git clone -b $(GOLLAMA_VERSION) --recurse-submodules https://github.com/go-skynet/go-llama.cpp go-llama
+	$(MAKE) -C go-llama libbinding.a
+
+go-llama-generic:
+	git clone -b $(GOLLAMA_VERSION) --recurse-submodules https://github.com/go-skynet/go-llama.cpp go-llama
+	$(MAKE) -C go-llama generic-libbinding.a
+
+prepare: go-llama go-gpt4all-j/libgptj.a
+	$(GOCMD) mod edit -replace github.com/go-skynet/go-llama.cpp=$(shell pwd)/go-llama
+	$(GOCMD) mod edit -replace github.com/go-skynet/go-gpt4all-j.cpp=$(shell pwd)/go-gpt4all-j
+
+prepare-generic: go-llama-generic go-gpt4all-j/libgptj.a-generic
+	$(GOCMD) mod edit -replace github.com/go-skynet/go-llama.cpp=$(shell pwd)/go-llama
+	$(GOCMD) mod edit -replace github.com/go-skynet/go-gpt4all-j.cpp=$(shell pwd)/go-gpt4all-j
+	
+clean: ## Remove build related file
+	rm -fr ./go-llama
+	rm -rf ./go-gpt4all-j
+	rm -rf $(BINARY_NAME)
+
+## Run:
+run: prepare
+	$(GOCMD) run ./ api
+
+## Test:
+test: ## Run the tests of the project
+	$(GOTEST) -v -race ./... $(OUTPUT_OPTIONS)
+
+## Help:
+help: ## Show this help.
+	@echo ''
+	@echo 'Usage:'
+	@echo '  ${YELLOW}make${RESET} ${GREEN}<target>${RESET}'
+	@echo ''
+	@echo 'Targets:'
+	@awk 'BEGIN {FS = ":.*?## "} { \
+		if (/^[a-zA-Z_-]+:.*?##.*$$/) {printf "    ${YELLOW}%-20s${GREEN}%s${RESET}\n", $$1, $$2} \
+		else if (/^## .*$$/) {printf "  ${CYAN}%s${RESET}\n", substr($$1,4)} \
+		}' $(MAKEFILE_LIST)
--- a/README.md
+++ b/README.md
@@ -1,15 +1,24 @@
 ## :camel: llama-cli


-llama-cli is a straightforward golang CLI interface and API compatible with OpenAI for [llama.cpp](https://github.com/ggerganov/llama.cpp), it supports multiple-models and also provides a simple command line interface that allows text generation using a GPT-based model like llama directly from the terminal. 
+llama-cli is a straightforward, drop-in replacement API compatible with OpenAI for local CPU inferencing, based on [llama.cpp](https://github.com/ggerganov/llama.cpp), [gpt4all](https://github.com/nomic-ai/gpt4all) and [ggml](https://github.com/ggerganov/ggml), including support GPT4ALL-J which is Apache 2.0 Licensed and can be used for commercial purposes.

-It is compatible with the models supported by `llama.cpp`. You might need to convert older models to the new format, see [here](https://github.com/ggerganov/llama.cpp#using-gpt4all) for instance to run `gpt4all`.
+- OpenAI compatible API
+- Supports multiple-models
+- Once loaded the first time, it keep models loaded in memory for faster inference
+- Provides a simple command line interface that allows text generation directly from the terminal
+- Support for prompt templates
+- Doesn't shell-out, but uses C bindings for a faster inference and better performance. Uses [go-llama.cpp](https://github.com/go-skynet/go-llama.cpp) and [go-gpt4all-j.cpp](https://github.com/go-skynet/go-gpt4all-j.cpp).

-`llama-cli` doesn't shell-out, it uses https://github.com/go-skynet/go-llama.cpp, which is a golang binding of [llama.cpp](https://github.com/ggerganov/llama.cpp).
+## Model compatibility
+
+It is compatible with the models supported by [llama.cpp](https://github.com/ggerganov/llama.cpp) and also [GPT4ALL-J](https://github.com/nomic-ai/gpt4all).
+
+Note: You might need to convert older models to the new format, see [here](https://github.com/ggerganov/llama.cpp#using-gpt4all) for instance to run `gpt4all`.

 ## Usage

-You can use `docker-compose`:
+The easiest way to run llama-cli is by using `docker-compose`:

 ```bash

@@ -19,8 +28,8 @@ cd llama-cli
 # copy your models to models/
 cp your-model.bin models/

-# (optional) Edit the .env file to set the number of concurrent threads used for inference
-# echo "THREADS=14" > .env
+# (optional) Edit the .env file to set things like context size and threads
+# vim .env

 # start with docker-compose
 docker compose up -d --build
@@ -28,16 +37,17 @@ docker compose up -d --build
 # Now API is accessible at localhost:8080
 curl http://localhost:8080/v1/models
 # {"object":"list","data":[{"id":"your-model.bin","object":"model"}]}
+
 curl http://localhost:8080/v1/completions -H "Content-Type: application/json" -d '{
     "model": "your-model.bin",            
     "prompt": "A long time ago in a galaxy far, far away",
     "temperature": 0.7
   }'
-
-
 ```

-Note: You can use a default template for every model in your model path, by creating a corresponding file with the `.tmpl` suffix next to your model. For instance, if the model is called `foo.bin`, you can create a sibiling file, `foo.bin.tmpl` which will be used as a default prompt, for instance this can be used with alpaca:
+Note: The API doesn't inject a default prompt for talking to the model, while the CLI does. You have to use a prompt similar to what's described in the standford-alpaca docs: https://github.com/tatsu-lab/stanford_alpaca#data-release.
+
+You can use a default template for every model present in your model path, by creating a corresponding file with the `.tmpl` suffix next to your model. For instance, if the model is called `foo.bin`, you can create a sibiling file, `foo.bin.tmpl` which will be used as a default prompt, for instance this can be used with alpaca:

 ```
 Below is an instruction that describes a task. Write a response that appropriately completes the request.
@@ -48,6 +58,8 @@ Below is an instruction that describes a task. Write a response that appropriate
 ### Response:
 ```

+See the [prompt-templates](https://github.com/go-skynet/llama-cli/tree/master/prompt-templates) directory in this repository for templates for most popular models.
+
 ## Container images

 `llama-cli` comes by default as a container image. You can check out all the available images with corresponding tags [here](https://quay.io/repository/go-skynet/llama-cli?tab=tags&tag=latest)
@@ -86,7 +98,7 @@ llama-cli --model <model_path> --instruction <instruction> [--input <input>] [--
 | template     | TEMPLATE             |               | A file containing a template for output formatting (optional).  |
 | instruction  | INSTRUCTION          |               | Input prompt text or instruction. "-" for STDIN.   |
 | input        | INPUT                | -             | Path to text or "-" for STDIN.                    |
-| model        | MODEL_PATH           |               | The path to the pre-trained GPT-based model.      |
+| model        | MODEL           |               | The path to the pre-trained GPT-based model.      |
 | tokens       | TOKENS               | 128           | The maximum number of tokens to generate. |
 | threads      | THREADS              | NumCPU()      | The number of threads to use for text generation. |
 | temperature  | TEMPERATURE          | 0.95          | Sampling temperature for model output. ( values between `0.1` and `1.0` )  |
@@ -184,22 +196,6 @@ You can list all the models available with:
 curl http://localhost:8080/v1/models
 ```

-## Web interface
-
-There is also available a simple web interface (for instance, http://localhost:8080/) which can be used as a playground.
-
-Note: The API doesn't inject a template for talking to the instance, while the CLI does. You have to use a prompt similar to what's described in the standford-alpaca docs: https://github.com/tatsu-lab/stanford_alpaca#data-release, for instance:
-
-```
-Below is an instruction that describes a task. Write a response that appropriately completes the request.
-
-### Instruction:
-{instruction}
-
-### Response:
-```
-
-
 ## Using other models

 gpt4all (https://github.com/nomic-ai/gpt4all) works as well, however the original model needs to be converted (same applies for old alpaca models, too):
@@ -214,32 +210,6 @@ python 828bddec6162a023114ce19146cb2b82/gistfile1.txt models tokenizer.model
 # There will be a new model with the ".tmp" extension, you have to use that one!
 ```

-### Golang client API
-
-The `llama-cli` codebase has also a small client in go that can be used alongside with the api:
-
-```golang
-package main
-
-import (
-	"fmt"
-
-	client "github.com/go-skynet/llama-cli/client"
-)
-
-func main() {
-
-	cli := client.NewClient("http://ip:port")
-
-	out, err := cli.Predict("What's an alpaca?")
-	if err != nil {
-		panic(err)
-	}
-
-	fmt.Println(out)
-}
-```
-
 ### Windows compatibility

 It should work, however you need to make sure you give enough resources to the container. See https://github.com/go-skynet/llama-cli/issues/2
--- a/api/api.go
+++ b/api/api.go
@@ -1,19 +1,16 @@
 package api

 import (
-	"embed"
 	"fmt"
-	"net/http"
-	"strconv"
 	"strings"
 	"sync"

 	model "github.com/go-skynet/llama-cli/pkg/model"

+	gptj "github.com/go-skynet/go-gpt4all-j.cpp"
 	llama "github.com/go-skynet/go-llama.cpp"
 	"github.com/gofiber/fiber/v2"
 	"github.com/gofiber/fiber/v2/middleware/cors"
-	"github.com/gofiber/fiber/v2/middleware/filesystem"
 	"github.com/gofiber/fiber/v2/middleware/recover"
 )

@@ -26,10 +23,10 @@ type OpenAIResponse struct {
 }

 type Choice struct {
-	Index        int     `json:"index,omitempty"`
-	FinishReason string  `json:"finish_reason,omitempty"`
-	Message      Message `json:"message,omitempty"`
-	Text         string  `json:"text,omitempty"`
+	Index        int      `json:"index,omitempty"`
+	FinishReason string   `json:"finish_reason,omitempty"`
+	Message      *Message `json:"message,omitempty"`
+	Text         string   `json:"text,omitempty"`
 }

 type Message struct {
@@ -47,24 +44,33 @@ type OpenAIRequest struct {

 	// Prompt is read only by completion API calls
 	Prompt string `json:"prompt"`
-	
+
 	// Messages is read only by chat/completion API calls
 	Messages []Message `json:"messages"`

+	Echo bool `json:"echo"`
 	// Common options between all the API calls
 	TopP        float64 `json:"top_p"`
 	TopK        int     `json:"top_k"`
 	Temperature float64 `json:"temperature"`
 	Maxtokens   int     `json:"max_tokens"`
+
+	N int `json:"n"`
+
+	// Custom parameters - not present in the OpenAI API
+	Batch     int  `json:"batch"`
+	F16       bool `json:"f16kv"`
+	IgnoreEOS bool `json:"ignore_eos"`
+
+	Seed int `json:"seed"`
 }

-//go:embed index.html
-var indexHTML embed.FS
-
-func openAIEndpoint(chat bool, defaultModel *llama.LLama, loader *model.ModelLoader, threads int, defaultMutex *sync.Mutex, mutexMap *sync.Mutex, mutexes map[string]*sync.Mutex) func(c *fiber.Ctx) error {
+// https://platform.openai.com/docs/api-reference/completions
+func openAIEndpoint(chat bool, loader *model.ModelLoader, threads, ctx int, f16 bool, defaultMutex *sync.Mutex, mutexMap *sync.Mutex, mutexes map[string]*sync.Mutex) func(c *fiber.Ctx) error {
 	return func(c *fiber.Ctx) error {
 		var err error
 		var model *llama.LLama
+		var gptModel *gptj.GPTJ

 		input := new(OpenAIRequest)
 		// Get input data from the request body
@@ -73,14 +79,24 @@ func openAIEndpoint(chat bool, defaultModel *llama.LLama, loader *model.ModelLoa
 		}

 		if input.Model == "" {
-			if defaultModel == nil {
-				return fmt.Errorf("no default model loaded, and no model specified")
-			}
-			model = defaultModel
+			return fmt.Errorf("no model specified")
 		} else {
-			model, err = loader.LoadModel(input.Model)
-			if err != nil {
-				return err
+			// Try to load the model with both
+			var llamaerr error
+			llamaOpts := []llama.ModelOption{}
+			if ctx != 0 {
+				llamaOpts = append(llamaOpts, llama.SetContext(ctx))
+			}
+			if f16 {
+				llamaOpts = append(llamaOpts, llama.EnableF16Memory)
+			}
+
+			model, llamaerr = loader.LoadLLaMAModel(input.Model, llamaOpts...)
+			if llamaerr != nil {
+				gptModel, err = loader.LoadGPTJModel(input.Model)
+				if err != nil {
+					return fmt.Errorf("llama: %s gpt: %s", llamaerr.Error(), err.Error()) // llama failed first, so we want to catch both errors
+				}
 			}
 		}

@@ -139,36 +155,102 @@ func openAIEndpoint(chat bool, defaultModel *llama.LLama, loader *model.ModelLoa
 			predInput = templatedInput
 		}

-		// Generate the prediction using the language model
-		prediction, err := model.Predict(
-			predInput,
-			llama.SetTemperature(temperature),
-			llama.SetTopP(topP),
-			llama.SetTopK(topK),
-			llama.SetTokens(tokens),
-			llama.SetThreads(threads),
-		)
-		if err != nil {
-			return err
+		result := []Choice{}
+
+		n := input.N
+
+		if input.N == 0 {
+			n = 1
 		}

-		if chat {
-			// Return the chat prediction in the response body
-			return c.JSON(OpenAIResponse{
-				Model:   input.Model,
-				Choices: []Choice{{Message: Message{Role: "assistant", Content: prediction}}},
-			})
+		var predFunc func() (string, error)
+		switch {
+		case gptModel != nil:
+			predFunc = func() (string, error) {
+				// Generate the prediction using the language model
+				predictOptions := []gptj.PredictOption{
+					gptj.SetTemperature(temperature),
+					gptj.SetTopP(topP),
+					gptj.SetTopK(topK),
+					gptj.SetTokens(tokens),
+					gptj.SetThreads(threads),
+				}
+
+				if input.Batch != 0 {
+					predictOptions = append(predictOptions, gptj.SetBatch(input.Batch))
+				}
+
+				if input.Seed != 0 {
+					predictOptions = append(predictOptions, gptj.SetSeed(input.Seed))
+				}
+
+				return gptModel.Predict(
+					predInput,
+					predictOptions...,
+				)
+			}
+		case model != nil:
+			predFunc = func() (string, error) {
+				// Generate the prediction using the language model
+				predictOptions := []llama.PredictOption{
+					llama.SetTemperature(temperature),
+					llama.SetTopP(topP),
+					llama.SetTopK(topK),
+					llama.SetTokens(tokens),
+					llama.SetThreads(threads),
+				}
+
+				if input.Batch != 0 {
+					predictOptions = append(predictOptions, llama.SetBatch(input.Batch))
+				}
+
+				if input.F16 {
+					predictOptions = append(predictOptions, llama.EnableF16KV)
+				}
+
+				if input.IgnoreEOS {
+					predictOptions = append(predictOptions, llama.IgnoreEOS)
+				}
+
+				if input.Seed != 0 {
+					predictOptions = append(predictOptions, llama.SetSeed(input.Seed))
+				}
+
+				return model.Predict(
+					predInput,
+					predictOptions...,
+				)
+			}
+		}
+
+		for i := 0; i < n; i++ {
+			var prediction string
+
+			prediction, err := predFunc()
+			if err != nil {
+				return err
+			}
+
+			if input.Echo {
+				prediction = predInput + prediction
+			}
+
+			if chat {
+				result = append(result, Choice{Message: &Message{Role: "assistant", Content: prediction}})
+			} else {
+				result = append(result, Choice{Text: prediction})
+			}
 		}

 		// Return the prediction in the response body
 		return c.JSON(OpenAIResponse{
 			Model:   input.Model,
-			Choices: []Choice{{Text: prediction}},
+			Choices: result,
 		})
 	}
 }

-func Start(defaultModel *llama.LLama, loader *model.ModelLoader, listenAddr string, threads int) error {
+func Start(loader *model.ModelLoader, listenAddr string, threads, ctxSize int, f16 bool) error {
 	app := fiber.New()

 	// Default middleware config
@@ -181,8 +263,8 @@ func Start(defaultModel *llama.LLama, loader *model.ModelLoader, listenAddr stri
 	var mumutex = &sync.Mutex{}

 	// openAI compatible API endpoint
-	app.Post("/v1/chat/completions", openAIEndpoint(true, defaultModel, loader, threads, mutex, mumutex, mu))
-	app.Post("/v1/completions", openAIEndpoint(false, defaultModel, loader, threads, mutex, mumutex, mu))
+	app.Post("/v1/chat/completions", openAIEndpoint(true, loader, threads, ctxSize, f16, mutex, mumutex, mu))
+	app.Post("/v1/completions", openAIEndpoint(false, loader, threads, ctxSize, f16, mutex, mumutex, mu))
 	app.Get("/v1/models", func(c *fiber.Ctx) error {
 		models, err := loader.ListModels()
 		if err != nil {
@@ -202,74 +284,6 @@ func Start(defaultModel *llama.LLama, loader *model.ModelLoader, listenAddr stri
 		})
 	})

-	app.Use("/", filesystem.New(filesystem.Config{
-		Root:         http.FS(indexHTML),
-		NotFoundFile: "index.html",
-	}))
-
-	/*
-		curl --location --request POST 'http://localhost:8080/predict' --header 'Content-Type: application/json' --data-raw '{
-		    "text": "What is an alpaca?",
-		    "topP": 0.8,
-		    "topK": 50,
-		    "temperature": 0.7,
-		    "tokens": 100
-		}'
-	*/
-	// Endpoint to generate the prediction
-	app.Post("/predict", func(c *fiber.Ctx) error {
-		mutex.Lock()
-		defer mutex.Unlock()
-		// Get input data from the request body
-		input := new(struct {
-			Text string `json:"text"`
-		})
-		if err := c.BodyParser(input); err != nil {
-			return err
-		}
-
-		// Set the parameters for the language model prediction
-		topP, err := strconv.ParseFloat(c.Query("topP", "0.9"), 64) // Default value of topP is 0.9
-		if err != nil {
-			return err
-		}
-
-		topK, err := strconv.Atoi(c.Query("topK", "40")) // Default value of topK is 40
-		if err != nil {
-			return err
-		}
-
-		temperature, err := strconv.ParseFloat(c.Query("temperature", "0.5"), 64) // Default value of temperature is 0.5
-		if err != nil {
-			return err
-		}
-
-		tokens, err := strconv.Atoi(c.Query("tokens", "128")) // Default value of tokens is 128
-		if err != nil {
-			return err
-		}
-
-		// Generate the prediction using the language model
-		prediction, err := defaultModel.Predict(
-			input.Text,
-			llama.SetTemperature(temperature),
-			llama.SetTopP(topP),
-			llama.SetTopK(topK),
-			llama.SetTokens(tokens),
-			llama.SetThreads(threads),
-		)
-		if err != nil {
-			return err
-		}
-
-		// Return the prediction in the response body
-		return c.JSON(struct {
-			Prediction string `json:"prediction"`
-		}{
-			Prediction: prediction,
-		})
-	})
-
 	// Start the server
 	app.Listen(listenAddr)
 	return nil
--- a/api/index.html
+++ b/api/index.html
@@ -1,120 +0,0 @@
-<!DOCTYPE html>
-<html>
-<head>
-    <title>llama-cli</title>
-    <meta charset="UTF-8">
-    <meta name="viewport" content="width=device-width, initial-scale=1">
-    <link rel="stylesheet" href="https://cdnjs.cloudflare.com/ajax/libs/font-awesome/5.15.3/css/all.min.css" crossorigin="anonymous" referrerpolicy="no-referrer" />
-    <link rel="stylesheet" href="https://stackpath.bootstrapcdn.com/bootstrap/4.3.1/css/bootstrap.min.css">
-</head>
-<style>
-    @keyframes rotating {
-    from {
-        transform: rotate(0deg);
-    }
-    to {
-        transform: rotate(360deg);
-    }
-}
-
-.waiting {
-    animation: rotating 1s linear infinite;
-}
-
-</style>
-<body>
-
-<div class="container mt-5" x-data="{ templates:[
-    {
-      name: 'Alpaca: Instruction without input',
-      text: `Below is an instruction that describes a task. Write a response that appropriately completes the request.
-
-### Instruction:
-{{.Instruction}}
-
-### Response:`,
-    },
-    {
-      name: 'Alpaca: Instruction with input',
-      text: `Below is an instruction that describes a task, paired with an input that provides further context. Write a response that appropriately completes the request.
-
-### Instruction:
-{{.Instruction}}
-
-### Input:
-{{.Input}}
-
-### Response:`,
-    }
-  ], selectedTemplate: '', selectedTemplateText: '' }">
-    <h1>llama-cli API</h1>
-    <div class="form-group">
-        <label for="inputText">Input Text:</label>
-        <textarea class="form-control" id="inputText" rows="6" placeholder="Your text input here..." x-text="selectedTemplateText"></textarea>
-    </div>
-    <div class="form-group">
-        <label for="templateSelect">Select Template:</label>
-        <select class="form-control" id="templateSelect" x-model="selectedTemplateText">
-            <option value="">None</option>
-            <template x-for="(template, index) in templates" :key="index">
-                <option :value="template.text" x-text="template.name"></option>
-            </template>
-        </select>
-    </div>
-    <div class="form-group">
-        <label for="topP">Top P:</label>
-        <input type="range" step="0.01" min="0" max="1" class="form-control" id="topP" value="0.20" name="topP" onchange="this.nextElementSibling.value = this.value" required>
-        <output>0.20</output>
-    </div>
-    <div class="form-group">
-        <label for="topK">Top K:</label>
-        <input type="number" class="form-control" id="topK" value="10000" name="topK"  required>
-    </div>
-    <div class="form-group">
-        <label for="temperature">Temperature:</label>
-        <input type="range" step="0.01" min="0" max="1" value="0.9" class="form-control" id="temperature" name="temperature" onchange="this.nextElementSibling.value = this.value"  required>
-        <output>0.9</output>
-    </div>
-    <div class="form-group">
-        <label for="tokens">Tokens:</label>
-        <input type="number" class="form-control" id="tokens" name="tokens" value="128" required>
-    </div>
-    <button class="btn btn-primary" x-on:click="submitRequest()">Submit <i class="fas fa-paper-plane"></i></button>
-    <hr>
-    <div class="form-group">
-        <label for="outputText">Output Text:</label>
-        <textarea class="form-control" id="outputText" rows="5" readonly></textarea>
-    </div>
-</div>
-
-<script defer src="https://cdn.jsdelivr.net/npm/alpinejs@3.x.x/dist/cdn.min.js"></script>
-<script>
-    function submitRequest() {
-        var button = document.querySelector("i.fa-paper-plane");
-        button.classList.add("waiting");
-        var text = document.getElementById("inputText").value;
-        var url = "/predict";
-        var data = {
-            "text": text,
-            "topP": document.getElementById("topP").value,
-            "topK": document.getElementById("topK").value,
-            "temperature": document.getElementById("temperature").value,
-            "tokens": document.getElementById("tokens").value
-        };
-        fetch(url, {
-            method: "POST",
-            headers: {
-                "Content-Type": "application/json"
-            },
-            body: JSON.stringify(data)
-        })
-        .then(response => response.json())
-        .then(data => {
-            document.getElementById("outputText").value = data.prediction;
-            button.classList.remove("waiting");
-        })
-        .catch(error => { console.error(error); button.classList.remove("waiting"); });
-    }
-</script>
-</body>
-</html>
--- a/client/client.go
+++ b/client/client.go
@@ -1,75 +0,0 @@
-package client
-
-import (
-	"bytes"
-	"encoding/json"
-	"fmt"
-	"net/http"
-)
-
-type Prediction struct {
-	Prediction string `json:"prediction"`
-}
-
-type Client struct {
-	baseURL  string
-	client   *http.Client
-	endpoint string
-}
-
-func NewClient(baseURL string) *Client {
-	return &Client{
-		baseURL:  baseURL,
-		client:   &http.Client{},
-		endpoint: "/predict",
-	}
-}
-
-type InputData struct {
-	Text        string  `json:"text"`
-	TopP        float64 `json:"topP,omitempty"`
-	TopK        int     `json:"topK,omitempty"`
-	Temperature float64 `json:"temperature,omitempty"`
-	Tokens      int     `json:"tokens,omitempty"`
-}
-
-func (c *Client) Predict(text string, opts ...InputOption) (string, error) {
-	input := NewInputData(opts...)
-	input.Text = text
-
-	// encode input data to JSON format
-	inputBytes, err := json.Marshal(input)
-	if err != nil {
-		return "", err
-	}
-
-	// create HTTP request
-	url := c.baseURL + c.endpoint
-	req, err := http.NewRequest("POST", url, bytes.NewBuffer(inputBytes))
-	if err != nil {
-		return "", err
-	}
-
-	// set request headers
-	req.Header.Set("Content-Type", "application/json")
-
-	// send request and get response
-	resp, err := c.client.Do(req)
-	if err != nil {
-		return "", err
-	}
-	defer resp.Body.Close()
-
-	if resp.StatusCode != http.StatusOK {
-		return "", fmt.Errorf("request failed with status %d", resp.StatusCode)
-	}
-
-	// decode response body to Prediction struct
-	var prediction Prediction
-	err = json.NewDecoder(resp.Body).Decode(&prediction)
-	if err != nil {
-		return "", err
-	}
-
-	return prediction.Prediction, nil
-}
--- a/client/options.go
+++ b/client/options.go
@@ -1,51 +0,0 @@
-package client
-
-import "net/http"
-
-type ClientOption func(c *Client)
-
-func WithHTTPClient(httpClient *http.Client) ClientOption {
-	return func(c *Client) {
-		c.client = httpClient
-	}
-}
-
-func WithEndpoint(endpoint string) ClientOption {
-	return func(c *Client) {
-		c.endpoint = endpoint
-	}
-}
-
-type InputOption func(d *InputData)
-
-func NewInputData(opts ...InputOption) *InputData {
-	data := &InputData{}
-	for _, opt := range opts {
-		opt(data)
-	}
-	return data
-}
-
-func WithTopP(topP float64) InputOption {
-	return func(d *InputData) {
-		d.TopP = topP
-	}
-}
-
-func WithTopK(topK int) InputOption {
-	return func(d *InputData) {
-		d.TopK = topK
-	}
-}
-
-func WithTemperature(temperature float64) InputOption {
-	return func(d *InputData) {
-		d.Temperature = temperature
-	}
-}
-
-func WithTokens(tokens int) InputOption {
-	return func(d *InputData) {
-		d.Tokens = tokens
-	}
-}
--- a/docker-compose.yaml
+++ b/docker-compose.yaml
@@ -1,15 +1,30 @@
 version: '3.6'

 services:
+
+  # chatgpt:
+  #   image: ghcr.io/mckaywrigley/chatbot-ui:main
+  #   # platform: linux/amd64
+  #   ports:
+  #     - 3000:3000
+  #   environment:
+  #     - 'OPENAI_API_KEY=sk-000000000000000'
+  #     - 'OPENAI_API_HOST=http://api:8080'
+
  api:
    image: quay.io/go-skynet/llama-cli:latest
-    build: .
-    volumes:
-      - ./models:/models
+    build:
+      context: .
+      dockerfile: Dockerfile
+      # args:
+        # BUILD_TYPE: generic # Uncomment to build CPU generic code that works on most HW
    ports:
      - 8080:8080
    environment:
-      - MODELS_PATH=/models
-      - CONTEXT_SIZE=700
+      - MODELS_PATH=$MODELS_PATH
+      - CONTEXT_SIZE=$CONTEXT_SIZE
      - THREADS=$THREADS
-    command: api
+    volumes:
+      - ./models:/models:cached
+    command: api
+    
--- a/go.mod
+++ b/go.mod
@@ -3,7 +3,7 @@ module github.com/go-skynet/llama-cli
 go 1.19

 require (
-	github.com/go-skynet/go-llama.cpp v0.0.0-20230415155049-9260bfd28bc4
+	github.com/go-skynet/go-llama.cpp v0.0.0-20230415213228-bac222030640
 	github.com/gofiber/fiber/v2 v2.42.0
 	github.com/urfave/cli/v2 v2.25.0
 )
@@ -11,6 +11,7 @@ require (
 require (
 	github.com/andybalholm/brotli v1.0.4 // indirect
 	github.com/cpuguy83/go-md2man/v2 v2.0.2 // indirect
+	github.com/go-skynet/go-gpt4all-j.cpp v0.0.0-20230419091210-303cf2a59a94 // indirect
 	github.com/google/uuid v1.3.0 // indirect
 	github.com/klauspost/compress v1.15.9 // indirect
 	github.com/mattn/go-colorable v0.1.13 // indirect
--- a/go.sum
+++ b/go.sum
@@ -3,8 +3,10 @@ github.com/andybalholm/brotli v1.0.4/go.mod h1:fO7iG3H7G2nSZ7m0zPUDn85XEX2GTukHG
 github.com/cpuguy83/go-md2man/v2 v2.0.2 h1:p1EgwI/C7NhT0JmVkwCD2ZBK8j4aeHQX2pMHHBfMQ6w=
 github.com/cpuguy83/go-md2man/v2 v2.0.2/go.mod h1:tgQtvFlXSQOSOSIRvRPT7W67SCa46tRHOmNcaadrF8o=
 github.com/go-logr/logr v1.2.3 h1:2DntVwHkVopvECVRSlL5PSo9eG+cAkDCuckLubN+rq0=
-github.com/go-skynet/go-llama.cpp v0.0.0-20230415155049-9260bfd28bc4 h1:u/y9MlPHOeIj636IQmrf9ptMjjdgCVIcsfb7lMFh39M=
-github.com/go-skynet/go-llama.cpp v0.0.0-20230415155049-9260bfd28bc4/go.mod h1:35AKIEMY+YTKCBJIa/8GZcNGJ2J+nQk1hQiWo/OnEWw=
+github.com/go-skynet/go-gpt4all-j.cpp v0.0.0-20230419091210-303cf2a59a94 h1:rtrrMvlIq+g0/ltXjDdLeNtz0uc4wJ4Qs15GFU4ba4c=
+github.com/go-skynet/go-gpt4all-j.cpp v0.0.0-20230419091210-303cf2a59a94/go.mod h1:5VZ9XbcINI0XcHhkcX8GPK8TplFGAzu1Hrg4tNiMCtI=
+github.com/go-skynet/go-llama.cpp v0.0.0-20230415213228-bac222030640 h1:8SSVbQ3yvq7JnfLCLF4USV0PkQnnduUkaNCv/hHDa3E=
+github.com/go-skynet/go-llama.cpp v0.0.0-20230415213228-bac222030640/go.mod h1:35AKIEMY+YTKCBJIa/8GZcNGJ2J+nQk1hQiWo/OnEWw=
 github.com/go-task/slim-sprig v0.0.0-20230315185526-52ccab3ef572 h1:tfuBGBXKqDEevZMzYi5KSi8KkcZtzBcTgAUUtapy0OI=
 github.com/gofiber/fiber/v2 v2.42.0 h1:Fnp7ybWvS+sjNQsFvkhf4G8OhXswvB6Vee8hM/LyS+8=
 github.com/gofiber/fiber/v2 v2.42.0/go.mod h1:3+SGNjqMh5VQH5Vz2Wdi43zTIV16ktlFd3x3R6O1Zlc=
--- a/main.go
+++ b/main.go
@@ -34,11 +34,6 @@ var nonEmptyInput string = `Below is an instruction that describes a task, paire
 ### Response:
 `

-func llamaFromOptions(ctx *cli.Context) (*llama.LLama, error) {
-	opts := []llama.ModelOption{llama.SetContext(ctx.Int("context-size"))}
-	return llama.New(ctx.String("model"), opts...)
-}
-
 func templateString(t string, in interface{}) (string, error) {
 	// Parse the template
 	tmpl, err := template.New("prompt").Parse(t)
@@ -57,7 +52,7 @@ func templateString(t string, in interface{}) (string, error) {
 var modelFlags = []cli.Flag{
 	&cli.StringFlag{
 		Name:    "model",
-		EnvVars: []string{"MODEL_PATH"},
+		EnvVars: []string{"MODEL"},
 	},
 	&cli.IntFlag{
 		Name:    "tokens",
@@ -125,6 +120,10 @@ echo "An Alpaca (Vicugna pacos) is a domesticated species of South American came

 				Name: "api",
 				Flags: []cli.Flag{
+					&cli.BoolFlag{
+						Name:    "f16",
+						EnvVars: []string{"F16"},
+					},
 					&cli.IntFlag{
 						Name:    "threads",
 						EnvVars: []string{"THREADS"},
@@ -134,10 +133,6 @@ echo "An Alpaca (Vicugna pacos) is a domesticated species of South American came
 						Name:    "models-path",
 						EnvVars: []string{"MODELS_PATH"},
 					},
-					&cli.StringFlag{
-						Name:    "default-model",
-						EnvVars: []string{"default-model"},
-					},
 					&cli.StringFlag{
 						Name:    "address",
 						EnvVars: []string{"ADDRESS"},
@@ -150,19 +145,7 @@ echo "An Alpaca (Vicugna pacos) is a domesticated species of South American came
 					},
 				},
 				Action: func(ctx *cli.Context) error {
-
-					var defaultModel *llama.LLama
-					defModel := ctx.String("default-model")
-					if defModel != "" {
-						opts := []llama.ModelOption{llama.SetContext(ctx.Int("context-size"))}
-						var err error
-						defaultModel, err = llama.New(ctx.String("default-model"), opts...)
-						if err != nil {
-							return err
-						}
-					}
-
-					return api.Start(defaultModel, model.NewModelLoader(ctx.String("models-path")), ctx.String("address"), ctx.Int("threads"))
+					return api.Start(model.NewModelLoader(ctx.String("models-path")), ctx.String("address"), ctx.Int("threads"), ctx.Int("context-size"), ctx.Bool("f16"))
 				},
 			},
 		},
@@ -217,7 +200,8 @@ echo "An Alpaca (Vicugna pacos) is a domesticated species of South American came
 				os.Exit(1)
 			}

-			l, err := llamaFromOptions(ctx)
+			opts := []llama.ModelOption{llama.SetContext(ctx.Int("context-size"))}
+			l, err := llama.New(ctx.String("model"), opts...)
 			if err != nil {
 				fmt.Println("Loading the model failed:", err.Error())
 				os.Exit(1)
--- a/pkg/model/loader.go
+++ b/pkg/model/loader.go
@@ -10,6 +10,7 @@ import (
 	"sync"
 	"text/template"

+	gptj "github.com/go-skynet/go-gpt4all-j.cpp"
 	llama "github.com/go-skynet/go-llama.cpp"
 )

@@ -17,11 +18,12 @@ type ModelLoader struct {
 	modelPath        string
 	mu               sync.Mutex
 	models           map[string]*llama.LLama
+	gptmodels        map[string]*gptj.GPTJ
 	promptsTemplates map[string]*template.Template
 }

 func NewModelLoader(modelPath string) *ModelLoader {
-	return &ModelLoader{modelPath: modelPath, models: make(map[string]*llama.LLama), promptsTemplates: make(map[string]*template.Template)}
+	return &ModelLoader{modelPath: modelPath, gptmodels: make(map[string]*gptj.GPTJ), models: make(map[string]*llama.LLama), promptsTemplates: make(map[string]*template.Template)}
 }

 func (ml *ModelLoader) ListModels() ([]string, error) {
@@ -62,16 +64,81 @@ func (ml *ModelLoader) TemplatePrefix(modelName string, in interface{}) (string,
 	return buf.String(), nil
 }

-func (ml *ModelLoader) LoadModel(modelName string, opts ...llama.ModelOption) (*llama.LLama, error) {
+func (ml *ModelLoader) loadTemplate(modelName, modelFile string) error {
+	modelTemplateFile := fmt.Sprintf("%s.tmpl", modelFile)
+
+	// Check if the model path exists
+	if _, err := os.Stat(modelTemplateFile); err != nil {
+		return nil
+	}
+
+	dat, err := os.ReadFile(modelTemplateFile)
+	if err != nil {
+		return err
+	}
+
+	// Parse the template
+	tmpl, err := template.New("prompt").Parse(string(dat))
+	if err != nil {
+		return err
+	}
+	ml.promptsTemplates[modelName] = tmpl
+
+	return nil
+}
+
+func (ml *ModelLoader) LoadGPTJModel(modelName string) (*gptj.GPTJ, error) {
 	ml.mu.Lock()
 	defer ml.mu.Unlock()

 	// Check if we already have a loaded model
 	modelFile := filepath.Join(ml.modelPath, modelName)

+	if m, ok := ml.gptmodels[modelFile]; ok {
+		return m, nil
+	}
+
+	// Check if the model path exists
+	if _, err := os.Stat(modelFile); os.IsNotExist(err) {
+		// try to find a s.bin
+		modelBin := fmt.Sprintf("%s.bin", modelFile)
+		if _, err := os.Stat(modelBin); os.IsNotExist(err) {
+			return nil, err
+		} else {
+			modelName = fmt.Sprintf("%s.bin", modelName)
+			modelFile = modelBin
+		}
+	}
+
+	// Load the model and keep it in memory for later use
+	model, err := gptj.New(modelFile)
+	if err != nil {
+		return nil, err
+	}
+
+	// If there is a prompt template, load it
+	if err := ml.loadTemplate(modelName, modelFile); err != nil {
+		return nil, err
+	}
+
+	ml.gptmodels[modelFile] = model
+	return model, err
+}
+
+func (ml *ModelLoader) LoadLLaMAModel(modelName string, opts ...llama.ModelOption) (*llama.LLama, error) {
+	ml.mu.Lock()
+	defer ml.mu.Unlock()
+
+	// Check if we already have a loaded model
+	modelFile := filepath.Join(ml.modelPath, modelName)
 	if m, ok := ml.models[modelFile]; ok {
 		return m, nil
 	}
+	// TODO: This needs refactoring, it's really bad to have it in here
+	// Check if we have a GPTJ model loaded instead
+	if _, ok := ml.gptmodels[modelFile]; ok {
+		return nil, fmt.Errorf("this model is a GPTJ one")
+	}

 	// Check if the model path exists
 	if _, err := os.Stat(modelFile); os.IsNotExist(err) {
@@ -92,21 +159,8 @@ func (ml *ModelLoader) LoadModel(modelName string, opts ...llama.ModelOption) (*
 	}

 	// If there is a prompt template, load it
-
-	modelTemplateFile := fmt.Sprintf("%s.tmpl", modelFile)
-	// Check if the model path exists
-	if _, err := os.Stat(modelTemplateFile); err == nil {
-		dat, err := os.ReadFile(modelTemplateFile)
-		if err != nil {
-			return nil, err
-		}
-
-		// Parse the template
-		tmpl, err := template.New("prompt").Parse(string(dat))
-		if err != nil {
-			return nil, err
-		}
-		ml.promptsTemplates[modelName] = tmpl
+	if err := ml.loadTemplate(modelName, modelFile); err != nil {
+		return nil, err
 	}

 	ml.models[modelFile] = model
--- a/prompt-templates/alpaca.tmpl
+++ b/prompt-templates/alpaca.tmpl
@@ -0,0 +1,6 @@
+Below is an instruction that describes a task. Write a response that appropriately completes the request.
+
+### Instruction:
+{{.Input}}
+
+### Response:
--- a/prompt-templates/ggml-gpt4all-j.tmpl
+++ b/prompt-templates/ggml-gpt4all-j.tmpl
@@ -0,0 +1,4 @@
+The prompt below is a question to answer, a task to complete, or a conversation to respond to; decide which and write an appropriate response.
+### Prompt:
+{{.Input}}
+### Response:
--- a/prompt-templates/koala.tmpl
+++ b/prompt-templates/koala.tmpl
@@ -0,0 +1 @@
+BEGINNING OF CONVERSATION: USER: {{.Input}} GPT:
--- a/prompt-templates/vicuna.tmpl
+++ b/prompt-templates/vicuna.tmpl
@@ -0,0 +1,6 @@
+Below is an instruction that describes a task. Write a response that appropriately completes the request.
+
+### Instruction:
+{{.Input}}
+
+### Response:
Author	SHA1	Message	Date
Ettore Di Giacinto	7fec26f5d3	Enhancements (#34 ) Signed-off-by: mudler <mudler@c3os.io>	2023-04-19 17:10:29 +02:00
Ettore Di Giacinto	a9a875ee2b	⬆️ Bump llama.cpp (#33 ) Signed-off-by: mudler <mudler@c3os.io>	2023-04-17 21:34:02 +02:00
Ettore Di Giacinto	db5ac715f3	Use a reasonable default context size (#31 )	2023-04-17 18:45:42 +02:00
Ettore Di Giacinto	0b330d90ad	feat: drop embedded webui (#27 ) Signed-off-by: mudler <mudler@c3os.io>	2023-04-16 10:46:20 +02:00
Ettore Di Giacinto	63601fabd1	feat: drop default model and llama-specific API (#26 ) Signed-off-by: mudler <mudler@c3os.io>	2023-04-16 10:40:50 +02:00
Ettore Di Giacinto	1370b4482f	📖 Add prompt-templates examples (#25 ) Signed-off-by: mudler <mudler@c3os.io>	2023-04-16 10:24:15 +02:00
Ettore Di Giacinto	b062f3142b	feat: enhance API, expose more parameters (#24 ) Signed-off-by: mudler <mudler@c3os.io>	2023-04-16 10:16:48 +02:00
Marc R Kellerman	c37175271f	feature: makefile & updates (#23 ) Co-authored-by: mudler <mudler@c3os.io> Co-authored-by: Ettore Di Giacinto <mudler@users.noreply.github.com>	2023-04-15 16:39:07 -07:00
				`@@ -0,0 +1 @@`
				`BEGINNING OF CONVERSATION: USER: {{.Input}} GPT:`