Update linux.md

Merge pull request #595 from jmorganca/mxyng/install.sh
ignore systemctl is-system-running exit code
2023-09-25 16:10:32 -07:00 · 2023-09-25 15:49:47 -07:00 · 2023-09-25 15:47:45 -07:00 · 2023-09-25 18:36:46 -04:00 · 2023-09-25 15:30:58 -07:00 · 2023-09-25 14:09:40 -07:00
166 changed files with 19586 additions and 7148 deletions
--- a/.dockerignore
+++ b/.dockerignore
@@ -1,7 +1,8 @@
-build
-llama/build
-.venv
 .vscode
 ollama
 app
-web
+dist
+scripts
+llm/llama.cpp/ggml
+llm/llama.cpp/gguf
+.env
--- a/.gitignore
+++ b/.gitignore
@@ -2,9 +2,7 @@
 .vscode
 .env
 .venv
-*.spec
-build
+.swp
 dist
-__pycache__
 ollama
 ggml-metal.metal
--- a/.gitmodules
+++ b/.gitmodules
@@ -0,0 +1,10 @@
+[submodule "llm/llama.cpp/ggml"]
+    path = llm/llama.cpp/ggml
+    url = https://github.com/ggerganov/llama.cpp.git
+    ignore = dirty
+    shallow = true
+[submodule "llm/llama.cpp/gguf"]
+    path = llm/llama.cpp/gguf
+    url = https://github.com/ggerganov/llama.cpp.git
+    ignore = dirty
+    shallow = true
--- a/36
+++ b/36
@@ -1,17 +1,31 @@
-FROM golang:1.20
-RUN apt-get update && apt-get install -y cmake
-WORKDIR /go/src/github.com/jmorganca/ollama
-COPY . .
-RUN cmake -S llama -B llama/build && cmake --build llama/build
-RUN CGO_ENABLED=1 go build -ldflags '-linkmode external -extldflags "-static"' .
+ARG CUDA_VERSION=12.2.0
+
+FROM nvidia/cuda:$CUDA_VERSION-devel-ubuntu22.04
+
+ARG TARGETARCH
+ARG VERSION=0.0.0
+
+WORKDIR /go/src/github.com/jmorganca/ollama
+RUN apt-get update && apt-get install -y git build-essential cmake
+ADD https://dl.google.com/go/go1.21.1.linux-$TARGETARCH.tar.gz /tmp/go1.21.1.tar.gz
+RUN mkdir -p /usr/local && tar xz -C /usr/local </tmp/go1.21.1.tar.gz
+
+COPY . .
+ENV GOARCH=$TARGETARCH
+RUN /usr/local/go/bin/go generate ./... \
+    && /usr/local/go/bin/go build -ldflags "-linkmode=external -extldflags='-static' -X=github.com/jmorganca/ollama/version.Version=$VERSION -X=github.com/jmorganca/ollama/server.mode=release" .
+
+FROM ubuntu:22.04
+ENV OLLAMA_HOST 0.0.0.0
+
+RUN apt-get update && apt-get install -y ca-certificates

-FROM alpine
-COPY --from=0 /go/src/github.com/jmorganca/ollama/ollama /bin/ollama
-EXPOSE 11434
 ARG USER=ollama
 ARG GROUP=ollama
-RUN addgroup -g 1000 $GROUP && adduser -u 1000 -DG $GROUP $USER
+RUN groupadd $GROUP && useradd -m -g $GROUP $USER
+
+COPY --from=0 /go/src/github.com/jmorganca/ollama/ollama /bin/ollama
+
 USER $USER:$GROUP
 ENTRYPOINT ["/bin/ollama"]
-ENV OLLAMA_HOST 0.0.0.0
 CMD ["serve"]
--- a/Dockerfile.build
+++ b/Dockerfile.build
@@ -0,0 +1,29 @@
+ARG VERSION=0.0.0
+
+# centos7 amd64 dependencies
+FROM --platform=linux/amd64 nvidia/cuda:11.8.0-devel-centos7 AS base-amd64
+RUN yum install -y https://repo.ius.io/ius-release-el7.rpm centos-release-scl && \
+    yum update -y && \
+    yum install -y devtoolset-10-gcc devtoolset-10-gcc-c++ git236 wget
+RUN wget "https://github.com/Kitware/CMake/releases/download/v3.27.6/cmake-3.27.6-linux-x86_64.sh" -O cmake-installer.sh && chmod +x cmake-installer.sh && ./cmake-installer.sh --skip-license --prefix=/usr/local
+ENV PATH /opt/rh/devtoolset-10/root/usr/bin:$PATH
+
+# centos8 arm64 dependencies
+FROM --platform=linux/arm64 nvidia/cuda:11.4.3-devel-centos8 AS base-arm64
+RUN sed -i -e 's/mirrorlist/#mirrorlist/g' -e 's|#baseurl=http://mirror.centos.org|baseurl=http://vault.centos.org|g' /etc/yum.repos.d/CentOS-*
+RUN yum install -y git cmake
+
+FROM base-${TARGETARCH}
+ARG TARGETARCH
+
+# install go
+ADD https://dl.google.com/go/go1.21.1.linux-$TARGETARCH.tar.gz /tmp/go1.21.1.tar.gz
+RUN mkdir -p /usr/local && tar xz -C /usr/local </tmp/go1.21.1.tar.gz
+
+# build the final binary
+WORKDIR /go/src/github.com/jmorganca/ollama
+COPY . .
+ENV GOARCH=$TARGETARCH
+
+RUN /usr/local/go/bin/go generate ./... && \
+    /usr/local/go/bin/go build -ldflags "-X=github.com/jmorganca/ollama/version.Version=$VERSION -X=github.com/jmorganca/ollama/server.mode=release" .
--- a/19
+++ b/19
@@ -1,19 +0,0 @@
-default: ollama
-
-.PHONY: llama
-llama:
-	cmake -S llama -B llama/build -DLLAMA_METAL=on
-	cmake --build llama/build
-
-.PHONY: ollama
-ollama: llama
-	go build .
-
-.PHONY: app
-app: ollama
-	npm install --prefix app
-	npm run --prefix app make:sign
-
-clean:
-	go clean
-	rm -rf llama/build
--- a/README.md
+++ b/README.md
@@ -1,109 +1,222 @@
-![ollama](https://github.com/jmorganca/ollama/assets/251292/961f99bb-251a-4eec-897d-1ba99997ad0f)
+<div align="center">
+  <picture>
+    <source media="(prefers-color-scheme: dark)" height="200px" srcset="https://github.com/jmorganca/ollama/assets/3325447/56ea1849-1284-4645-8970-956de6e51c3c">
+    <img alt="logo" height="200px" src="https://github.com/jmorganca/ollama/assets/3325447/0d0b44e2-8f4a-4e99-9b52-a5c1c741c8f7">
+  </picture>
+</div>

 # Ollama

-Run large language models with `llama.cpp`.
+[![Discord](https://dcbadge.vercel.app/api/server/ollama?style=flat&compact=true)](https://discord.gg/ollama)

-> Note: certain models that can be run with Ollama are intended for research and/or non-commercial use only.
+Run, create, and share large language models (LLMs).

-### Features
+> Note: Ollama is in early preview. Please report any issues you find.

- Download and run popular large language models
- Switch between multiple models on the fly
- Hardware acceleration where available (Metal, CUDA)
- Fast inference server written in Go, powered by [llama.cpp](https://github.com/ggerganov/llama.cpp)
- REST API to use with your application (python, typescript SDKs coming soon)
-
-## Install
+## Download

 - [Download](https://ollama.ai/download) for macOS
- Download for Windows (coming soon)
- Docker: `docker run -p 11434:11434 ollama/ollama`
-
-You can also build the [binary from source](#building).
+- Download for Windows and Linux (coming soon)
+- Build [from source](#building)

 ## Quickstart

-Run a fast and simple model.
+To run and chat with [Llama 2](https://ai.meta.com/llama), the new model by Meta:

 ```
-ollama run orca
+ollama run llama2
 ```

-## Example models
+## Model library

-### 💬 Chat
+Ollama supports a list of open-source models available on [ollama.ai/library](https://ollama.ai/library 'ollama model library')

-Have a conversation.
+Here are some example open-source models that can be downloaded:
+
+| Model                    | Parameters | Size  | Download                        |
+| ------------------------ | ---------- | ----- | ------------------------------- |
+| Llama2                   | 7B         | 3.8GB | `ollama pull llama2`            |
+| Llama2 13B               | 13B        | 7.3GB | `ollama pull llama2:13b`        |
+| Llama2 70B               | 70B        | 39GB  | `ollama pull llama2:70b`        |
+| Llama2 Uncensored        | 7B         | 3.8GB | `ollama pull llama2-uncensored` |
+| Code Llama               | 7B         | 3.8GB | `ollama pull codellama`         |
+| Orca Mini                | 3B         | 1.9GB | `ollama pull orca-mini`         |
+| Vicuna                   | 7B         | 3.8GB | `ollama pull vicuna`            |
+| Nous-Hermes              | 7B         | 3.8GB | `ollama pull nous-hermes`       |
+| Nous-Hermes 13B          | 13B        | 7.3GB | `ollama pull nous-hermes:13b`   |
+| Wizard Vicuna Uncensored | 13B        | 7.3GB | `ollama pull wizard-vicuna`     |
+
+> Note: You should have at least 8 GB of RAM to run the 3B models, 16 GB to run the 7B models, and 32 GB to run the 13B models.
+
+## Examples
+
+### Pull a public model

 ```
-ollama run vicuna "Why is the sky blue?"
+ollama pull llama2
 ```

-### 🗺️ Instructions
+> This command can also be used to update a local model. Only updated changes will be pulled.

-Get a helping hand.
+### Run a model interactively

 ```
-ollama run orca "Write an email to my boss."
+ollama run llama2
+>>> hi
+Hello! How can I help you today?
 ```

-### 🔎 Ask questions about documents
-
-Send the contents of a document and ask questions about it.
+For multiline input, you can wrap text with `"""`:

 ```
-ollama run nous-hermes "$(cat input.txt)", please summarize this story
+>>> """Hello,
+... world!
+... """
+I'm a basic program that prints the famous "Hello, world!" message to the console.
 ```

-### 📖 Storytelling
-
-Venture into the unknown.
+### Run a model non-interactively

 ```
-ollama run nous-hermes "Once upon a time"
+$ ollama run llama2 'tell me a joke'
+ Sure! Here's a quick one:
+ Why did the scarecrow win an award? Because he was outstanding in his field!
 ```

-## Advanced usage
+```
+$ cat <<EOF >prompts.txt
+tell me a joke about llamas
+tell me another one
+EOF
+$ ollama run llama2 <prompts.txt
+>>> tell me a joke about llamas
+ Why did the llama refuse to play hide-and-seek?
+ nobody likes to be hided!

-### Run a local model
+>>> tell me another one
+ Sure, here's another one:
+
+Why did the llama go to the bar?
+To have a hay-often good time!
+```
+
+### Run a model on contents of a text file

 ```
-ollama run ~/Downloads/vicuna-7b-v1.3.ggmlv3.q4_1.bin
+$ ollama run llama2 "summarize this file:" "$(cat README.md)"
+ Ollama is a lightweight, extensible framework for building and running language models on the local machine. It provides a simple API for creating, running, and managing models, as well as a library of pre-built models that can be easily used in a variety of applications.
 ```

+### Customize a model
+
+Pull a base model:
+
+```
+ollama pull llama2
+```
+
+Create a `Modelfile`:
+
+```
+FROM llama2
+
+# set the temperature to 1 [higher is more creative, lower is more coherent]
+PARAMETER temperature 1
+
+# set the system prompt
+SYSTEM """
+You are Mario from Super Mario Bros. Answer as Mario, the assistant, only.
+"""
+```
+
+Next, create and run the model:
+
+```
+ollama create mario -f ./Modelfile
+ollama run mario
+>>> hi
+Hello! It's your friend Mario.
+```
+
+For more examples, see the [examples](./examples) directory. For more information on creating a Modelfile, see the [Modelfile](./docs/modelfile.md) documentation.
+
+### Listing local models
+
+```
+ollama list
+```
+
+### Removing local models
+
+```
+ollama rm llama2
+```
+
+## Model packages
+
+### Overview
+
+Ollama bundles model weights, configurations, and data into a single package, defined by a [Modelfile](./docs/modelfile.md).
+
+<picture>
+  <source media="(prefers-color-scheme: dark)" height="480" srcset="https://github.com/jmorganca/ollama/assets/251292/2fd96b5f-191b-45c1-9668-941cfad4eb70">
+  <img alt="logo" height="480" src="https://github.com/jmorganca/ollama/assets/251292/2fd96b5f-191b-45c1-9668-941cfad4eb70">
+</picture>
+
 ## Building

-```
-make
-```
-
-To run it start the server:
+Install `cmake` and `go`:

 ```
-./ollama server &
+brew install cmake
+brew install go
 ```

-Finally, run a model!
+Then generate dependencies and build:

 ```
-./ollama run ~/Downloads/vicuna-7b-v1.3.ggmlv3.q4_1.bin
+go generate ./...
+go build .
 ```

-## API Reference
-
-### `POST /api/pull`
-
-Download a model
+Next, start the server:

 ```
-curl -X POST http://localhost:11343/api/pull -d '{"model": "orca"}'
+./ollama serve
 ```

-### `POST /api/generate`
-
-Complete a prompt
+Finally, in a separate shell, run a model:

 ```
-curl -X POST http://localhost:11434/api/generate -d '{"model": "orca", "prompt": "hello!", "stream": true}'
+./ollama run llama2
 ```
+
+## REST API
+
+> See the [API documentation](./docs/api.md) for all endpoints.
+
+Ollama has an API for running and managing models. For example to generate text from a model:
+
+```
+curl -X POST http://localhost:11434/api/generate -d '{
+  "model": "llama2",
+  "prompt":"Why is the sky blue?"
+}'
+```
+
+## Community Projects using Ollama
+
+| Project                                                                    | Description                                                                                                                                                  |
+| -------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ |
+| [LangChain][1] and [LangChain.js][2]                                       | Also, there is a question-answering [example][3].                                                                                                            |
+| [Continue](https://github.com/continuedev/continue)                        | Embeds Ollama inside Visual Studio Code. The extension lets you highlight code to add to the prompt, ask questions in the sidebar, and generate code inline. |
+| [LiteLLM](https://github.com/BerriAI/litellm)                              | Lightweight Python package to simplify LLM API calls.                                                                                                        |
+| [Discord AI Bot](https://github.com/mekb-turtle/discord-ai-bot)            | Interact with Ollama as a chatbot on Discord.                                                                                                                |
+| [Raycast Ollama](https://github.com/MassimilianoPasquini97/raycast_ollama) | Raycast extension to use Ollama for local llama inference on Raycast.                                                                                        |
+| [Simple HTML UI](https://github.com/rtcfirefly/ollama-ui)                  | Also, there is a Chrome extension.                                                                                                                           |
+| [Ollama-GUI](https://github.com/ollama-interface/Ollama-Gui?tab=readme-ov-file)                  | 🖥️ Mac Chat Interface ⚡️                                                                                                                           |
+| [Emacs client](https://github.com/zweifisch/ollama)                        |                                                                                                                                                              |
+
+[1]: https://python.langchain.com/docs/integrations/llms/ollama
+[2]: https://js.langchain.com/docs/modules/model_io/models/llms/integrations/ollama
+[3]: https://js.langchain.com/docs/use_cases/question_answering/local_retrieval_qa
--- a/api/client.go
+++ b/api/client.go
@@ -6,26 +6,124 @@ import (
 	"context"
 	"encoding/json"
 	"fmt"
+	"io"
 	"net/http"
 	"net/url"
+	"os"
+	"runtime"
+	"strings"
+
+	"github.com/jmorganca/ollama/version"
+)
+
+const DefaultHost = "127.0.0.1:11434"
+
+var (
+	envHost = os.Getenv("OLLAMA_HOST")
 )

 type Client struct {
-	base url.URL
+	Base    url.URL
+	HTTP    http.Client
+	Headers http.Header
 }

-func NewClient(hosts ...string) *Client {
-	host := "127.0.0.1:11434"
-	if len(hosts) > 0 {
-		host = hosts[0]
+func checkError(resp *http.Response, body []byte) error {
+	if resp.StatusCode < http.StatusBadRequest {
+		return nil
 	}

-	return &Client{
-		base: url.URL{Scheme: "http", Host: host},
+	apiError := StatusError{StatusCode: resp.StatusCode}
+
+	err := json.Unmarshal(body, &apiError)
+	if err != nil {
+		// Use the full body as the message if we fail to decode a response.
+		apiError.ErrorMessage = string(body)
 	}
+
+	return apiError
 }

-func (c *Client) stream(ctx context.Context, method, path string, data any, callback func([]byte) error) error {
+// Host returns the default host to use for the client. It is determined in the following order:
+// 1. The OLLAMA_HOST environment variable
+// 2. The default host (localhost:11434)
+func Host() string {
+	if envHost != "" {
+		return envHost
+	}
+	return DefaultHost
+}
+
+// FromEnv creates a new client using Host() as the host. An error is returns
+// if the host is invalid.
+func FromEnv() (*Client, error) {
+	h := Host()
+	if !strings.HasPrefix(h, "http://") && !strings.HasPrefix(h, "https://") {
+		h = "http://" + h
+	}
+
+	u, err := url.Parse(h)
+	if err != nil {
+		return nil, fmt.Errorf("could not parse host: %w", err)
+	}
+
+	if u.Port() == "" {
+		u.Host += ":11434"
+	}
+
+	return &Client{Base: *u, HTTP: http.Client{}}, nil
+}
+
+func (c *Client) do(ctx context.Context, method, path string, reqData, respData any) error {
+	var reqBody io.Reader
+	var data []byte
+	var err error
+	if reqData != nil {
+		data, err = json.Marshal(reqData)
+		if err != nil {
+			return err
+		}
+		reqBody = bytes.NewReader(data)
+	}
+
+	requestURL := c.Base.JoinPath(path)
+	request, err := http.NewRequestWithContext(ctx, method, requestURL.String(), reqBody)
+	if err != nil {
+		return err
+	}
+
+	request.Header.Set("Content-Type", "application/json")
+	request.Header.Set("Accept", "application/json")
+	request.Header.Set("User-Agent", fmt.Sprintf("ollama/%s (%s %s) Go/%s", version.Version, runtime.GOARCH, runtime.GOOS, runtime.Version()))
+
+	for k, v := range c.Headers {
+		request.Header[k] = v
+	}
+
+	respObj, err := c.HTTP.Do(request)
+	if err != nil {
+		return err
+	}
+	defer respObj.Body.Close()
+
+	respBody, err := io.ReadAll(respObj.Body)
+	if err != nil {
+		return err
+	}
+
+	if err := checkError(respObj, respBody); err != nil {
+		return err
+	}
+
+	if len(respBody) > 0 && respData != nil {
+		if err := json.Unmarshal(respBody, respData); err != nil {
+			return err
+		}
+	}
+	return nil
+}
+
+func (c *Client) stream(ctx context.Context, method, path string, data any, fn func([]byte) error) error {
 	var buf *bytes.Buffer
 	if data != nil {
 		bts, err := json.Marshal(data)
@@ -36,13 +134,15 @@ func (c *Client) stream(ctx context.Context, method, path string, data any, call
 		buf = bytes.NewBuffer(bts)
 	}

-	request, err := http.NewRequestWithContext(ctx, method, c.base.JoinPath(path).String(), buf)
+	requestURL := c.Base.JoinPath(path)
+	request, err := http.NewRequestWithContext(ctx, method, requestURL.String(), buf)
 	if err != nil {
 		return err
 	}

 	request.Header.Set("Content-Type", "application/json")
 	request.Header.Set("Accept", "application/json")
+	request.Header.Set("User-Agent", fmt.Sprintf("ollama/%s (%s %s) Go/%s", version.Version, runtime.GOARCH, runtime.GOOS, runtime.Version()))

 	response, err := http.DefaultClient.Do(request)
 	if err != nil {
@@ -53,7 +153,7 @@ func (c *Client) stream(ctx context.Context, method, path string, data any, call
 	scanner := bufio.NewScanner(response.Body)
 	for scanner.Scan() {
 		var errorResponse struct {
-			Error string `json:"error"`
+			Error string `json:"error,omitempty"`
 		}

 		bts := scanner.Bytes()
@@ -61,11 +161,19 @@ func (c *Client) stream(ctx context.Context, method, path string, data any, call
 			return fmt.Errorf("unmarshal: %w", err)
 		}

-		if len(errorResponse.Error) > 0 {
-			return fmt.Errorf("stream: %s", errorResponse.Error)
+		if errorResponse.Error != "" {
+			return fmt.Errorf(errorResponse.Error)
 		}

-		if err := callback(bts); err != nil {
+		if response.StatusCode >= http.StatusBadRequest {
+			return StatusError{
+				StatusCode:   response.StatusCode,
+				Status:       response.Status,
+				ErrorMessage: errorResponse.Error,
+			}
+		}
+
+		if err := fn(bts); err != nil {
 			return err
 		}
 	}
@@ -86,11 +194,11 @@ func (c *Client) Generate(ctx context.Context, req *GenerateRequest, fn Generate
 	})
 }

-type PullProgressFunc func(PullProgress) error
+type PullProgressFunc func(ProgressResponse) error

 func (c *Client) Pull(ctx context.Context, req *PullRequest, fn PullProgressFunc) error {
 	return c.stream(ctx, http.MethodPost, "/api/pull", req, func(bts []byte) error {
-		var resp PullProgress
+		var resp ProgressResponse
 		if err := json.Unmarshal(bts, &resp); err != nil {
 			return err
 		}
@@ -98,3 +206,66 @@ func (c *Client) Pull(ctx context.Context, req *PullRequest, fn PullProgressFunc
 		return fn(resp)
 	})
 }
+
+type PushProgressFunc func(ProgressResponse) error
+
+func (c *Client) Push(ctx context.Context, req *PushRequest, fn PushProgressFunc) error {
+	return c.stream(ctx, http.MethodPost, "/api/push", req, func(bts []byte) error {
+		var resp ProgressResponse
+		if err := json.Unmarshal(bts, &resp); err != nil {
+			return err
+		}
+
+		return fn(resp)
+	})
+}
+
+type CreateProgressFunc func(ProgressResponse) error
+
+func (c *Client) Create(ctx context.Context, req *CreateRequest, fn CreateProgressFunc) error {
+	return c.stream(ctx, http.MethodPost, "/api/create", req, func(bts []byte) error {
+		var resp ProgressResponse
+		if err := json.Unmarshal(bts, &resp); err != nil {
+			return err
+		}
+
+		return fn(resp)
+	})
+}
+
+func (c *Client) List(ctx context.Context) (*ListResponse, error) {
+	var lr ListResponse
+	if err := c.do(ctx, http.MethodGet, "/api/tags", nil, &lr); err != nil {
+		return nil, err
+	}
+	return &lr, nil
+}
+
+func (c *Client) Copy(ctx context.Context, req *CopyRequest) error {
+	if err := c.do(ctx, http.MethodPost, "/api/copy", req, nil); err != nil {
+		return err
+	}
+	return nil
+}
+
+func (c *Client) Delete(ctx context.Context, req *DeleteRequest) error {
+	if err := c.do(ctx, http.MethodDelete, "/api/delete", req, nil); err != nil {
+		return err
+	}
+	return nil
+}
+
+func (c *Client) Show(ctx context.Context, req *ShowRequest) (*ShowResponse, error) {
+	var resp ShowResponse
+	if err := c.do(ctx, http.MethodPost, "/api/show", req, &resp); err != nil {
+		return nil, err
+	}
+	return &resp, nil
+}
+
+func (c *Client) Heartbeat(ctx context.Context) error {
+	if err := c.do(ctx, http.MethodHead, "/", nil, nil); err != nil {
+		return err
+	}
+	return nil
+}
--- a/api/client.py
+++ b/api/client.py
@@ -0,0 +1,225 @@
+import os
+import json
+import requests
+
+BASE_URL = os.environ.get('OLLAMA_HOST', 'http://localhost:11434')
+
+# Generate a response for a given prompt with a provided model. This is a streaming endpoint, so will be a series of responses.
+# The final response object will include statistics and additional data from the request. Use the callback function to override
+# the default handler.
+def generate(model_name, prompt, system=None, template=None, context=None, options=None, callback=None):
+    try:
+        url = f"{BASE_URL}/api/generate"
+        payload = {
+            "model": model_name, 
+            "prompt": prompt, 
+            "system": system, 
+            "template": template, 
+            "context": context, 
+            "options": options
+        }
+        
+        # Remove keys with None values
+        payload = {k: v for k, v in payload.items() if v is not None}
+        
+        with requests.post(url, json=payload, stream=True) as response:
+            response.raise_for_status()
+            
+            # Creating a variable to hold the context history of the final chunk
+            final_context = None
+            
+            # Variable to hold concatenated response strings if no callback is provided
+            full_response = ""
+
+            # Iterating over the response line by line and displaying the details
+            for line in response.iter_lines():
+                if line:
+                    # Parsing each line (JSON chunk) and extracting the details
+                    chunk = json.loads(line)
+                    
+                    # If a callback function is provided, call it with the chunk
+                    if callback:
+                        callback(chunk)
+                    else:
+                        # If this is not the last chunk, add the "response" field value to full_response and print it
+                        if not chunk.get("done"):
+                            response_piece = chunk.get("response", "")
+                            full_response += response_piece
+                            print(response_piece, end="", flush=True)
+                    
+                    # Check if it's the last chunk (done is true)
+                    if chunk.get("done"):
+                        final_context = chunk.get("context")
+            
+            # Return the full response and the final context
+            return full_response, final_context
+    except requests.exceptions.RequestException as e:
+        print(f"An error occurred: {e}")
+        return None, None
+
+# Create a model from a Modelfile. Use the callback function to override the default handler.
+def create(model_name, model_path, callback=None):
+    try:
+        url = f"{BASE_URL}/api/create"
+        payload = {"name": model_name, "path": model_path}
+        
+        # Making a POST request with the stream parameter set to True to handle streaming responses
+        with requests.post(url, json=payload, stream=True) as response:
+            response.raise_for_status()
+
+            # Iterating over the response line by line and displaying the status
+            for line in response.iter_lines():
+                if line:
+                    # Parsing each line (JSON chunk) and extracting the status
+                    chunk = json.loads(line)
+
+                    if callback:
+                        callback(chunk)
+                    else:
+                        print(f"Status: {chunk.get('status')}")
+    except requests.exceptions.RequestException as e:
+        print(f"An error occurred: {e}")
+
+# Pull a model from a the model registry. Cancelled pulls are resumed from where they left off, and multiple
+# calls to will share the same download progress. Use the callback function to override the default handler.
+def pull(model_name, insecure=False, callback=None):
+    try:
+        url = f"{BASE_URL}/api/pull"
+        payload = {
+            "name": model_name,
+            "insecure": insecure
+        }
+
+        # Making a POST request with the stream parameter set to True to handle streaming responses
+        with requests.post(url, json=payload, stream=True) as response:
+            response.raise_for_status()
+
+            # Iterating over the response line by line and displaying the details
+            for line in response.iter_lines():
+                if line:
+                    # Parsing each line (JSON chunk) and extracting the details
+                    chunk = json.loads(line)
+
+                    # If a callback function is provided, call it with the chunk
+                    if callback:
+                        callback(chunk)
+                    else:
+                        # Print the status message directly to the console
+                        print(chunk.get('status', ''), end='', flush=True)
+                    
+                    # If there's layer data, you might also want to print that (adjust as necessary)
+                    if 'digest' in chunk:
+                        print(f" - Digest: {chunk['digest']}", end='', flush=True)
+                        print(f" - Total: {chunk['total']}", end='', flush=True)
+                        print(f" - Completed: {chunk['completed']}", end='\n', flush=True)
+                    else:
+                        print()
+    except requests.exceptions.RequestException as e:
+        print(f"An error occurred: {e}")
+
+# Push a model to the model registry. Use the callback function to override the default handler.
+def push(model_name, insecure=False, callback=None):
+    try:
+        url = f"{BASE_URL}/api/push"
+        payload = {
+            "name": model_name,
+            "insecure": insecure
+        }
+
+        # Making a POST request with the stream parameter set to True to handle streaming responses
+        with requests.post(url, json=payload, stream=True) as response:
+            response.raise_for_status()
+
+            # Iterating over the response line by line and displaying the details
+            for line in response.iter_lines():
+                if line:
+                    # Parsing each line (JSON chunk) and extracting the details
+                    chunk = json.loads(line)
+
+                    # If a callback function is provided, call it with the chunk
+                    if callback:
+                        callback(chunk)
+                    else:
+                        # Print the status message directly to the console
+                        print(chunk.get('status', ''), end='', flush=True)
+                    
+                    # If there's layer data, you might also want to print that (adjust as necessary)
+                    if 'digest' in chunk:
+                        print(f" - Digest: {chunk['digest']}", end='', flush=True)
+                        print(f" - Total: {chunk['total']}", end='', flush=True)
+                        print(f" - Completed: {chunk['completed']}", end='\n', flush=True)
+                    else:
+                        print()
+    except requests.exceptions.RequestException as e:
+        print(f"An error occurred: {e}")
+
+# List models that are available locally.
+def list():
+    try:
+        response = requests.get(f"{BASE_URL}/api/tags")
+        response.raise_for_status()
+        data = response.json()
+        models = data.get('models', [])
+        return models
+
+    except requests.exceptions.RequestException as e:
+        print(f"An error occurred: {e}")
+        return None
+
+# Copy a model. Creates a model with another name from an existing model.
+def copy(source, destination):
+    try:
+        # Create the JSON payload
+        payload = {
+            "source": source,
+            "destination": destination
+        }
+        
+        response = requests.post(f"{BASE_URL}/api/copy", json=payload)
+        response.raise_for_status()
+        
+        # If the request was successful, return a message indicating that the copy was successful
+        return "Copy successful"
+
+    except requests.exceptions.RequestException as e:
+        print(f"An error occurred: {e}")
+        return None
+
+# Delete a model and its data.
+def delete(model_name):
+    try:
+        url = f"{BASE_URL}/api/delete"
+        payload = {"name": model_name}
+        response = requests.delete(url, json=payload)
+        response.raise_for_status()
+        return "Delete successful"
+    except requests.exceptions.RequestException as e:
+        print(f"An error occurred: {e}")
+        return None
+
+# Show info about a model.
+def show(model_name):
+    try:
+        url = f"{BASE_URL}/api/show"
+        payload = {"name": model_name}
+        response = requests.post(url, json=payload)
+        response.raise_for_status()
+        
+        # Parse the JSON response and return it
+        data = response.json()
+        return data
+    except requests.exceptions.RequestException as e:
+        print(f"An error occurred: {e}")
+        return None
+
+def heartbeat():
+    try:
+        url = f"{BASE_URL}/"
+        response = requests.head(url)
+        response.raise_for_status()
+        return "Ollama is running"
+    except requests.exceptions.RequestException as e:
+        print(f"An error occurred: {e}")
+        return "Ollama is not running"
+
+
--- a/api/types.go
+++ b/api/types.go
@@ -1,106 +1,349 @@
 package api

-type PullRequest struct {
-	Model string `json:"model"`
+import (
+	"encoding/json"
+	"fmt"
+	"log"
+	"math"
+	"os"
+	"reflect"
+	"strings"
+	"time"
+)
+
+type StatusError struct {
+	StatusCode   int
+	Status       string
+	ErrorMessage string `json:"error"`
 }

-type PullProgress struct {
-	Total     int64   `json:"total"`
-	Completed int64   `json:"completed"`
-	Percent   float64 `json:"percent"`
+func (e StatusError) Error() string {
+	switch {
+	case e.Status != "" && e.ErrorMessage != "":
+		return fmt.Sprintf("%s: %s", e.Status, e.ErrorMessage)
+	case e.Status != "":
+		return e.Status
+	case e.ErrorMessage != "":
+		return e.ErrorMessage
+	default:
+		// this should not happen
+		return "something went wrong, please see the ollama server logs for details"
+	}
 }

 type GenerateRequest struct {
+	Model    string `json:"model"`
+	Prompt   string `json:"prompt"`
+	System   string `json:"system"`
+	Template string `json:"template"`
+	Context  []int  `json:"context,omitempty"`
+
+	Options map[string]interface{} `json:"options"`
+}
+
+type EmbeddingRequest struct {
 	Model  string `json:"model"`
 	Prompt string `json:"prompt"`

-	ModelOptions   *ModelOptions   `json:"model_opts,omitempty"`
-	PredictOptions *PredictOptions `json:"predict_opts,omitempty"`
+	Options map[string]interface{} `json:"options"`
 }

-type ModelOptions struct {
-	ContextSize int    `json:"context_size,omitempty"`
-	Seed        int    `json:"seed,omitempty"`
-	NBatch      int    `json:"n_batch,omitempty"`
-	F16Memory   bool   `json:"memory_f16,omitempty"`
-	MLock       bool   `json:"mlock,omitempty"`
-	MMap        bool   `json:"mmap,omitempty"`
-	VocabOnly   bool   `json:"vocab_only,omitempty"`
-	LowVRAM     bool   `json:"low_vram,omitempty"`
-	Embeddings  bool   `json:"embeddings,omitempty"`
-	NUMA        bool   `json:"numa,omitempty"`
-	NGPULayers  int    `json:"gpu_layers,omitempty"`
-	MainGPU     string `json:"main_gpu,omitempty"`
-	TensorSplit string `json:"tensor_split,omitempty"`
+type EmbeddingResponse struct {
+	Embedding []float64 `json:"embedding"`
 }

-type PredictOptions struct {
-	Seed        int     `json:"seed,omitempty"`
-	Threads     int     `json:"threads,omitempty"`
-	Tokens      int     `json:"tokens,omitempty"`
-	TopK        int     `json:"top_k,omitempty"`
-	Repeat      int     `json:"repeat,omitempty"`
-	Batch       int     `json:"batch,omitempty"`
-	NKeep       int     `json:"nkeep,omitempty"`
-	TopP        float64 `json:"top_p,omitempty"`
-	Temperature float64 `json:"temp,omitempty"`
-	Penalty     float64 `json:"penalty,omitempty"`
-	F16KV       bool
-	DebugMode   bool
-	StopPrompts []string
-	IgnoreEOS   bool `json:"ignore_eos,omitempty"`
-
-	TailFreeSamplingZ float64 `json:"tfs_z,omitempty"`
-	TypicalP          float64 `json:"typical_p,omitempty"`
-	FrequencyPenalty  float64 `json:"freq_penalty,omitempty"`
-	PresencePenalty   float64 `json:"pres_penalty,omitempty"`
-	Mirostat          int     `json:"mirostat,omitempty"`
-	MirostatETA       float64 `json:"mirostat_lr,omitempty"`
-	MirostatTAU       float64 `json:"mirostat_ent,omitempty"`
-	PenalizeNL        bool    `json:"penalize_nl,omitempty"`
-	LogitBias         string  `json:"logit_bias,omitempty"`
-
-	PathPromptCache string
-	MLock           bool `json:"mlock,omitempty"`
-	MMap            bool `json:"mmap,omitempty"`
-	PromptCacheAll  bool
-	PromptCacheRO   bool
-	MainGPU         string
-	TensorSplit     string
+type CreateRequest struct {
+	Name string `json:"name"`
+	Path string `json:"path"`
 }

-var DefaultModelOptions ModelOptions = ModelOptions{
-	ContextSize: 512,
-	Seed:        0,
-	F16Memory:   true,
-	MLock:       false,
-	Embeddings:  true,
-	MMap:        true,
-	LowVRAM:     false,
+type DeleteRequest struct {
+	Name string `json:"name"`
 }

-var DefaultPredictOptions PredictOptions = PredictOptions{
-	Seed:              -1,
-	Threads:           -1,
-	Tokens:            512,
-	Penalty:           1.1,
-	Repeat:            64,
-	Batch:             512,
-	NKeep:             64,
-	TopK:              90,
-	TopP:              0.86,
-	TailFreeSamplingZ: 1.0,
-	TypicalP:          1.0,
-	Temperature:       0.8,
-	FrequencyPenalty:  0.0,
-	PresencePenalty:   0.0,
-	Mirostat:          0,
-	MirostatTAU:       5.0,
-	MirostatETA:       0.1,
-	MMap:              true,
-	StopPrompts:       []string{"llama"},
+type ShowRequest struct {
+	Name string `json:"name"`
+}
+
+type ShowResponse struct {
+	License    string `json:"license,omitempty"`
+	Modelfile  string `json:"modelfile,omitempty"`
+	Parameters string `json:"parameters,omitempty"`
+	Template   string `json:"template,omitempty"`
+	System     string `json:"system,omitempty"`
+}
+
+type CopyRequest struct {
+	Source      string `json:"source"`
+	Destination string `json:"destination"`
+}
+
+type PullRequest struct {
+	Name     string `json:"name"`
+	Insecure bool   `json:"insecure,omitempty"`
+	Username string `json:"username"`
+	Password string `json:"password"`
+}
+
+type ProgressResponse struct {
+	Status    string `json:"status"`
+	Digest    string `json:"digest,omitempty"`
+	Total     int    `json:"total,omitempty"`
+	Completed int    `json:"completed,omitempty"`
+}
+
+type PushRequest struct {
+	Name     string `json:"name"`
+	Insecure bool   `json:"insecure,omitempty"`
+	Username string `json:"username"`
+	Password string `json:"password"`
+}
+
+type ListResponse struct {
+	Models []ModelResponse `json:"models"`
+}
+
+type ModelResponse struct {
+	Name       string    `json:"name"`
+	ModifiedAt time.Time `json:"modified_at"`
+	Size       int       `json:"size"`
+	Digest     string    `json:"digest"`
+}
+
+type TokenResponse struct {
+	Token string `json:"token"`
 }

 type GenerateResponse struct {
-	Response string `json:"response"`
+	Model     string    `json:"model"`
+	CreatedAt time.Time `json:"created_at"`
+	Response  string    `json:"response,omitempty"`
+
+	Done    bool  `json:"done"`
+	Context []int `json:"context,omitempty"`
+
+	TotalDuration      time.Duration `json:"total_duration,omitempty"`
+	LoadDuration       time.Duration `json:"load_duration,omitempty"`
+	PromptEvalCount    int           `json:"prompt_eval_count,omitempty"`
+	PromptEvalDuration time.Duration `json:"prompt_eval_duration,omitempty"`
+	EvalCount          int           `json:"eval_count,omitempty"`
+	EvalDuration       time.Duration `json:"eval_duration,omitempty"`
+}
+
+func (r *GenerateResponse) Summary() {
+	if r.TotalDuration > 0 {
+		fmt.Fprintf(os.Stderr, "total duration:       %v\n", r.TotalDuration)
+	}
+
+	if r.LoadDuration > 0 {
+		fmt.Fprintf(os.Stderr, "load duration:        %v\n", r.LoadDuration)
+	}
+
+	if r.PromptEvalCount > 0 {
+		fmt.Fprintf(os.Stderr, "prompt eval count:    %d token(s)\n", r.PromptEvalCount)
+	}
+
+	if r.PromptEvalDuration > 0 {
+		fmt.Fprintf(os.Stderr, "prompt eval duration: %s\n", r.PromptEvalDuration)
+		fmt.Fprintf(os.Stderr, "prompt eval rate:     %.2f tokens/s\n", float64(r.PromptEvalCount)/r.PromptEvalDuration.Seconds())
+	}
+
+	if r.EvalCount > 0 {
+		fmt.Fprintf(os.Stderr, "eval count:           %d token(s)\n", r.EvalCount)
+	}
+
+	if r.EvalDuration > 0 {
+		fmt.Fprintf(os.Stderr, "eval duration:        %s\n", r.EvalDuration)
+		fmt.Fprintf(os.Stderr, "eval rate:            %.2f tokens/s\n", float64(r.EvalCount)/r.EvalDuration.Seconds())
+	}
+}
+
+type Options struct {
+	Seed int `json:"seed,omitempty"`
+
+	// Backend options
+	UseNUMA bool `json:"numa,omitempty"`
+
+	// Model options
+	NumCtx             int     `json:"num_ctx,omitempty"`
+	NumKeep            int     `json:"num_keep,omitempty"`
+	NumBatch           int     `json:"num_batch,omitempty"`
+	NumGQA             int     `json:"num_gqa,omitempty"`
+	NumGPU             int     `json:"num_gpu,omitempty"`
+	MainGPU            int     `json:"main_gpu,omitempty"`
+	LowVRAM            bool    `json:"low_vram,omitempty"`
+	F16KV              bool    `json:"f16_kv,omitempty"`
+	LogitsAll          bool    `json:"logits_all,omitempty"`
+	VocabOnly          bool    `json:"vocab_only,omitempty"`
+	UseMMap            bool    `json:"use_mmap,omitempty"`
+	UseMLock           bool    `json:"use_mlock,omitempty"`
+	EmbeddingOnly      bool    `json:"embedding_only,omitempty"`
+	RopeFrequencyBase  float32 `json:"rope_frequency_base,omitempty"`
+	RopeFrequencyScale float32 `json:"rope_frequency_scale,omitempty"`
+
+	// Predict options
+	NumPredict       int      `json:"num_predict,omitempty"`
+	TopK             int      `json:"top_k,omitempty"`
+	TopP             float32  `json:"top_p,omitempty"`
+	TFSZ             float32  `json:"tfs_z,omitempty"`
+	TypicalP         float32  `json:"typical_p,omitempty"`
+	RepeatLastN      int      `json:"repeat_last_n,omitempty"`
+	Temperature      float32  `json:"temperature,omitempty"`
+	RepeatPenalty    float32  `json:"repeat_penalty,omitempty"`
+	PresencePenalty  float32  `json:"presence_penalty,omitempty"`
+	FrequencyPenalty float32  `json:"frequency_penalty,omitempty"`
+	Mirostat         int      `json:"mirostat,omitempty"`
+	MirostatTau      float32  `json:"mirostat_tau,omitempty"`
+	MirostatEta      float32  `json:"mirostat_eta,omitempty"`
+	PenalizeNewline  bool     `json:"penalize_newline,omitempty"`
+	Stop             []string `json:"stop,omitempty"`
+
+	NumThread int `json:"num_thread,omitempty"`
+}
+
+func (opts *Options) FromMap(m map[string]interface{}) error {
+	valueOpts := reflect.ValueOf(opts).Elem() // names of the fields in the options struct
+	typeOpts := reflect.TypeOf(opts).Elem()   // types of the fields in the options struct
+
+	// build map of json struct tags to their types
+	jsonOpts := make(map[string]reflect.StructField)
+	for _, field := range reflect.VisibleFields(typeOpts) {
+		jsonTag := strings.Split(field.Tag.Get("json"), ",")[0]
+		if jsonTag != "" {
+			jsonOpts[jsonTag] = field
+		}
+	}
+
+	for key, val := range m {
+		if opt, ok := jsonOpts[key]; ok {
+			field := valueOpts.FieldByName(opt.Name)
+			if field.IsValid() && field.CanSet() {
+				if val == nil {
+					continue
+				}
+
+				switch field.Kind() {
+				case reflect.Int:
+					switch t := val.(type) {
+					case int64:
+						field.SetInt(t)
+					case float64:
+						// when JSON unmarshals numbers, it uses float64, not int
+						field.SetInt(int64(t))
+					default:
+						log.Printf("could not convert model parameter %v to int, skipped", key)
+					}
+				case reflect.Bool:
+					val, ok := val.(bool)
+					if !ok {
+						log.Printf("could not convert model parameter %v to bool, skipped", key)
+						continue
+					}
+					field.SetBool(val)
+				case reflect.Float32:
+					// JSON unmarshals to float64
+					val, ok := val.(float64)
+					if !ok {
+						log.Printf("could not convert model parameter %v to float32, skipped", key)
+						continue
+					}
+					field.SetFloat(val)
+				case reflect.String:
+					val, ok := val.(string)
+					if !ok {
+						log.Printf("could not convert model parameter %v to string, skipped", key)
+						continue
+					}
+					field.SetString(val)
+				case reflect.Slice:
+					// JSON unmarshals to []interface{}, not []string
+					val, ok := val.([]interface{})
+					if !ok {
+						log.Printf("could not convert model parameter %v to slice, skipped", key)
+						continue
+					}
+					// convert []interface{} to []string
+					slice := make([]string, len(val))
+					for i, item := range val {
+						str, ok := item.(string)
+						if !ok {
+							log.Printf("could not convert model parameter %v to slice of strings, skipped", key)
+							continue
+						}
+						slice[i] = str
+					}
+					field.Set(reflect.ValueOf(slice))
+				default:
+					return fmt.Errorf("unknown type loading config params: %v", field.Kind())
+				}
+			}
+		}
+	}
+	return nil
+}
+
+func DefaultOptions() Options {
+	return Options{
+		Seed: -1,
+
+		UseNUMA: false,
+
+		NumCtx:             2048,
+		NumKeep:            -1,
+		NumBatch:           512,
+		NumGPU:             -1, // -1 here indicates that NumGPU should be set dynamically
+		NumGQA:             1,
+		LowVRAM:            false,
+		F16KV:              true,
+		UseMMap:            true,
+		UseMLock:           false,
+		RopeFrequencyBase:  10000.0,
+		RopeFrequencyScale: 1.0,
+		EmbeddingOnly:      true,
+
+		RepeatLastN:      64,
+		RepeatPenalty:    1.1,
+		FrequencyPenalty: 0.0,
+		PresencePenalty:  0.0,
+		Temperature:      0.8,
+		TopK:             40,
+		TopP:             0.9,
+		TFSZ:             1.0,
+		TypicalP:         1.0,
+		Mirostat:         0,
+		MirostatTau:      5.0,
+		MirostatEta:      0.1,
+		PenalizeNewline:  true,
+
+		NumThread: 0, // let the runtime decide
+	}
+}
+
+type Duration struct {
+	time.Duration
+}
+
+func (d *Duration) UnmarshalJSON(b []byte) (err error) {
+	var v any
+	if err := json.Unmarshal(b, &v); err != nil {
+		return err
+	}
+
+	d.Duration = 5 * time.Minute
+
+	switch t := v.(type) {
+	case float64:
+		if t < 0 {
+			t = math.MaxFloat64
+		}
+
+		d.Duration = time.Duration(t)
+	case string:
+		d.Duration, err = time.ParseDuration(t)
+		if err != nil {
+			return err
+		}
+	}
+
+	return nil
 }
--- a/app/README.md
+++ b/app/README.md
@@ -1,7 +1,5 @@
 # Desktop

-_Note: the Ollama desktop app is a work in progress and is not ready yet for general use._
-
 This app builds upon Ollama to provide a desktop experience for running models.

 ## Developing
@@ -9,19 +7,15 @@ This app builds upon Ollama to provide a desktop experience for running models.
 First, build the `ollama` binary:

 ```
-make -C ..
+cd ..
+go build .
 ```

 Then run the desktop app with `npm start`:

 ```
+cd app
 npm install
 npm start
 ```

-## Coming soon
-
- Browse the latest available models on Hugging Face and other sources
- Keep track of previous conversations with models
- Switch quickly between models
- Connect to remote Ollama servers to run models
--- a/app/assets/iconDarkTemplate.png
+++ b/app/assets/iconDarkTemplate.png
--- a/app/assets/iconDarkTemplate@2x.png
+++ b/app/assets/iconDarkTemplate@2x.png
--- a/app/assets/iconDarkUpdateTemplate.png
+++ b/app/assets/iconDarkUpdateTemplate.png
--- a/app/assets/iconDarkUpdateTemplate@2x.png
+++ b/app/assets/iconDarkUpdateTemplate@2x.png
--- a/app/assets/iconTemplate.png
+++ b/app/assets/iconTemplate.png
--- a/app/assets/iconTemplate@2x.png
+++ b/app/assets/iconTemplate@2x.png
--- a/app/assets/iconUpdateTemplate.png
+++ b/app/assets/iconUpdateTemplate.png
--- a/app/assets/iconUpdateTemplate@2x.png
+++ b/app/assets/iconUpdateTemplate@2x.png
--- a/app/assets/ollama_icon_16x16Template.png
+++ b/app/assets/ollama_icon_16x16Template.png
--- a/app/assets/ollama_icon_16x16Template@2x.png
+++ b/app/assets/ollama_icon_16x16Template@2x.png
--- a/app/forge.config.ts
+++ b/app/forge.config.ts
@@ -1,4 +1,4 @@
-import type { ForgeConfig, ResolvedForgeConfig, ForgeMakeResult } from '@electron-forge/shared-types'
+import type { ForgeConfig } from '@electron-forge/shared-types'
 import { MakerSquirrel } from '@electron-forge/maker-squirrel'
 import { MakerZIP } from '@electron-forge/maker-zip'
 import { PublisherGithub } from '@electron-forge/publisher-github'
@@ -18,10 +18,15 @@ const config: ForgeConfig = {
    asar: true,
    icon: './assets/icon.icns',
    extraResource: [
-      '../ollama',
-      path.join(__dirname, './assets/ollama_icon_16x16Template.png'),
-      path.join(__dirname, './assets/ollama_icon_16x16Template@2x.png'),
-      ...(process.platform === 'darwin' ? ['../ggml-metal.metal'] : []),
+      '../dist/ollama',
+      path.join(__dirname, './assets/iconTemplate.png'),
+      path.join(__dirname, './assets/iconTemplate@2x.png'),
+      path.join(__dirname, './assets/iconUpdateTemplate.png'),
+      path.join(__dirname, './assets/iconUpdateTemplate@2x.png'),
+      path.join(__dirname, './assets/iconDarkTemplate.png'),
+      path.join(__dirname, './assets/iconDarkTemplate@2x.png'),
+      path.join(__dirname, './assets/iconDarkUpdateTemplate.png'),
+      path.join(__dirname, './assets/iconDarkUpdateTemplate@2x.png'),
    ],
    ...(process.env.SIGN
      ? {
@@ -36,6 +41,9 @@ const config: ForgeConfig = {
          },
        }
      : {}),
+    osxUniversal: {
+      x64ArchFiles: '**/ollama',
+    },
  },
  rebuildConfig: {},
  makers: [new MakerSquirrel({}), new MakerZIP({}, ['darwin'])],
@@ -58,7 +66,7 @@ const config: ForgeConfig = {
    new AutoUnpackNativesPlugin({}),
    new WebpackPlugin({
      mainConfig,
-      devContentSecurityPolicy: `default-src * 'unsafe-eval' 'unsafe-inline'`,
+      devContentSecurityPolicy: `default-src * 'unsafe-eval' 'unsafe-inline'; img-src data: 'self'`,
      renderer: {
        config: rendererConfig,
        nodeIntegration: true,
--- a/app/package-lock.json
+++ b/app/package-lock.json
--- a/app/package.json
+++ b/app/package.json
@@ -6,12 +6,14 @@
  "main": ".webpack/main",
  "scripts": {
    "start": "electron-forge start",
-    "package": "electron-forge package",
-    "package:sign": "SIGN=1 electron-forge package",
-    "make": "electron-forge make",
-    "make:sign": "SIGN=1 electron-forge make",
+    "package": "electron-forge package --arch universal",
+    "package:sign": "SIGN=1 electron-forge package --arch universal",
+    "make": "electron-forge make --arch universal",
+    "make:sign": "SIGN=1 electron-forge make --arch universal",
    "publish": "SIGN=1 electron-forge publish",
-    "lint": "eslint --ext .ts,.tsx ."
+    "lint": "eslint --ext .ts,.tsx .",
+    "format": "prettier --check . --ignore-path .gitignore",
+    "format:fix": "prettier --write . --ignore-path .gitignore"
  },
  "keywords": [],
  "author": {
@@ -30,6 +32,8 @@
    "@electron-forge/plugin-auto-unpack-natives": "^6.2.1",
    "@electron-forge/plugin-webpack": "^6.2.1",
    "@electron-forge/publisher-github": "^6.2.1",
+    "@electron/universal": "^1.4.1",
+    "@svgr/webpack": "^8.0.1",
    "@types/chmodr": "^1.0.0",
    "@types/node": "^20.4.0",
    "@types/react": "^18.2.14",
@@ -54,21 +58,27 @@
    "prettier": "^2.8.8",
    "prettier-plugin-tailwindcss": "^0.3.0",
    "style-loader": "^3.3.3",
+    "svg-inline-loader": "^0.8.2",
    "tailwindcss": "^3.3.2",
    "ts-loader": "^9.4.3",
    "ts-node": "^10.9.1",
    "typescript": "~4.5.4",
+    "url-loader": "^4.1.1",
    "webpack": "^5.88.0",
    "webpack-cli": "^5.1.4",
    "webpack-dev-server": "^4.15.1"
  },
  "dependencies": {
    "@electron/remote": "^2.0.10",
+    "@heroicons/react": "^2.0.18",
    "@segment/analytics-node": "^1.0.0",
+    "copy-to-clipboard": "^3.3.3",
    "electron-squirrel-startup": "^1.0.0",
    "electron-store": "^8.1.0",
    "react": "^18.2.0",
    "react-dom": "^18.2.0",
-    "uuid": "^9.0.0"
+    "uuid": "^9.0.0",
+    "winston": "^3.10.0",
+    "winston-daily-rotate-file": "^4.7.1"
  }
 }
--- a/app/src/app.css
+++ b/app/src/app.css
@@ -11,6 +11,10 @@ body {
  -webkit-app-region: drag;
 }

+.no-drag {
+  -webkit-app-region: no-drag;
+}
+
 .blink {
  -webkit-animation: 1s blink step-end infinite;
  -moz-animation: 1s blink step-end infinite;
--- a/app/src/app.tsx
+++ b/app/src/app.tsx
@@ -1,158 +1,120 @@
 import { useState } from 'react'
-import path from 'path'
-import os from 'os'
-import { dialog, getCurrentWindow } from '@electron/remote'
+import copy from 'copy-to-clipboard'
+import { CheckIcon, DocumentDuplicateIcon } from '@heroicons/react/24/outline'
+import Store from 'electron-store'
+import { getCurrentWindow, app } from '@electron/remote'

-const API_URL = 'http://127.0.0.1:7734'
+import { install } from './install'
+import OllamaIcon from './ollama.svg'

-type Message = {
-  sender: 'bot' | 'human'
-  content: string
-}
+const store = new Store()

-const userInfo = os.userInfo()
-
-async function generate(prompt: string, model: string, callback: (res: string) => void) {
-  const result = await fetch(`${API_URL}/generate`, {
-    method: 'POST',
-    headers: {
-      'Content-Type': 'application/json',
-    },
-    body: JSON.stringify({
-      prompt,
-      model,
-    }),
-  })
-
-  if (!result.ok) {
-    return
-  }
-
-  let reader = result.body.getReader()
-
-  while (true) {
-    const { done, value } = await reader.read()
-
-    if (done) {
-      break
-    }
-
-    let decoder = new TextDecoder()
-    let str = decoder.decode(value)
-
-    let re = /}\s*{/g
-    str = '[' + str.replace(re, '},{') + ']'
-    let messages = JSON.parse(str)
-
-    for (const message of messages) {
-      const choice = message.choices[0]
-
-      callback(choice.text)
-
-      if (choice.finish_reason === 'stop') {
-        break
-      }
-    }
-  }
-
-  return
+enum Step {
+  WELCOME = 0,
+  CLI,
+  FINISH,
 }

 export default function () {
-  const [prompt, setPrompt] = useState('')
-  const [messages, setMessages] = useState<Message[]>([])
-  const [model, setModel] = useState('')
-  const [generating, setGenerating] = useState(false)
+  const [step, setStep] = useState<Step>(Step.WELCOME)
+  const [commandCopied, setCommandCopied] = useState<boolean>(false)
+
+  const command = 'ollama run llama2'

  return (
-    <div className='flex min-h-screen flex-1 flex-col justify-between bg-white'>
-      <header className='drag sticky top-0 z-50 flex h-14 w-full flex-row items-center border-b border-black/10 bg-white/75 backdrop-blur-md'>
-        <div className='mx-auto w-full max-w-xl leading-none'>
-          <h1 className='text-sm font-medium'>{path.basename(model).replace('.bin', '')}</h1>
-        </div>
-      </header>
-      {model ? (
-        <section className='mx-auto mb-10 w-full max-w-xl flex-1 break-words'>
-          {messages.map((m, i) => (
-            <div className='my-4 flex gap-4' key={i}>
-              <div className='flex-none pr-1 text-lg'>
-                {m.sender === 'human' ? (
-                  <div className='mt-px flex h-6 w-6 items-center justify-center rounded-md bg-neutral-200 text-sm text-neutral-700'>
-                    {userInfo.username[0].toUpperCase()}
-                  </div>
-                ) : (
-                  <div className='mt-0.5 flex h-6 w-6 items-center justify-center rounded-md bg-blue-600 text-sm text-white'>
-                    {path.basename(model)[0].toUpperCase()}
-                  </div>
-                )}
-              </div>
-              <div className='flex-1 text-gray-800'>
-                {m.content}
-                {m.sender === 'bot' && generating && i === messages.length - 1 && (
-                  <span className='blink relative -top-[3px] left-1 text-[10px]'>█</span>
-                )}
+    <div className='drag'>
+      <div className='mx-auto flex min-h-screen w-full flex-col justify-between bg-white px-4 pt-16'>
+        {step === Step.WELCOME && (
+          <>
+            <div className='mx-auto text-center'>
+              <h1 className='mb-6 mt-4 text-2xl tracking-tight text-gray-900'>Welcome to Ollama</h1>
+              <p className='mx-auto w-[65%] text-sm text-gray-400'>
+                Let's get you up and running with your own large language models.
+              </p>
+              <button
+                onClick={() => setStep(Step.CLI)}
+                className='no-drag rounded-dm mx-auto my-8 w-[40%] rounded-md bg-black px-4 py-2 text-sm text-white hover:brightness-110'
+              >
+                Next
+              </button>
+            </div>
+            <div className='mx-auto'>
+              <OllamaIcon />
+            </div>
+          </>
+        )}
+        {step === Step.CLI && (
+          <>
+            <div className='mx-auto flex flex-col space-y-28 text-center'>
+              <h1 className='mt-4 text-2xl tracking-tight text-gray-900'>Install the command line</h1>
+              <pre className='mx-auto text-4xl text-gray-400'>&gt; ollama</pre>
+              <div className='mx-auto'>
+                <button
+                  onClick={async () => {
+                    try {
+                      await install()
+                      setStep(Step.FINISH)
+                    } catch (e) {
+                      console.error('could not install: ', e)
+                    } finally {
+                      getCurrentWindow().show()
+                      getCurrentWindow().focus()
+                    }
+                  }}
+                  className='no-drag rounded-dm mx-auto w-[60%] rounded-md bg-black px-4 py-2 text-sm text-white hover:brightness-110'
+                >
+                  Install
+                </button>
+                <p className='mx-auto my-4 w-[70%] text-xs text-gray-400'>
+                  You will be prompted for administrator access
+                </p>
              </div>
            </div>
-          ))}
-        </section>
-      ) : (
-        <section className='flex flex-1 select-none flex-col items-center justify-center pb-20'>
-          <h2 className='text-3xl font-light text-neutral-400'>No model selected</h2>
-          <button
-            onClick={async () => {
-              const res = await dialog.showOpenDialog(getCurrentWindow(), {
-                properties: ['openFile', 'multiSelections'],
-              })
-              if (res.canceled) {
-                return
-              }
-
-              setModel(res.filePaths[0])
-            }}
-            className='rounded-dm my-8 rounded-md bg-blue-600 px-4 py-2 text-sm text-white hover:brightness-110'
-          >
-            Open file...
-          </button>
-        </section>
-      )}
-      <div className='sticky bottom-0 bg-gradient-to-b from-transparent to-white'>
-        {model && (
-          <textarea
-            autoFocus
-            rows={1}
-            value={prompt}
-            placeholder='Send a message...'
-            onChange={e => setPrompt(e.target.value)}
-            className='mx-auto my-4 block w-full max-w-xl resize-none rounded-xl border border-gray-200 px-5 py-3.5 text-[15px] shadow-lg shadow-black/5 focus:outline-none'
-            onKeyDownCapture={async e => {
-              if (e.key === 'Enter' && !e.shiftKey) {
-                e.preventDefault()
-
-                if (generating) {
-                  return
-                }
-
-                if (!prompt) {
-                  return
-                }
-
-                await setMessages(messages => {
-                  return [...messages, { sender: 'human', content: prompt }, { sender: 'bot', content: '' }]
-                })
-
-                setPrompt('')
-
-                setGenerating(true)
-                await generate(prompt, model, res => {
-                  setMessages(messages => {
-                    let last = messages[messages.length - 1]
-                    return [...messages.slice(0, messages.length - 1), { ...last, content: last.content + res }]
-                  })
-                })
-                setGenerating(false)
-              }
-            }}
-          ></textarea>
+          </>
+        )}
+        {step === Step.FINISH && (
+          <>
+            <div className='mx-auto flex flex-col space-y-20 text-center'>
+              <h1 className='mt-4 text-2xl tracking-tight text-gray-900'>Run your first model</h1>
+              <div className='flex flex-col'>
+                <div className='group relative flex items-center'>
+                  <pre className='language-none text-2xs w-full rounded-md bg-gray-100 px-4 py-3 text-start leading-normal'>
+                    {command}
+                  </pre>
+                  <button
+                    className={`no-drag absolute right-[5px] px-2 py-2 ${
+                      commandCopied
+                        ? 'text-gray-900 opacity-100 hover:cursor-auto'
+                        : 'text-gray-200 opacity-50 hover:cursor-pointer'
+                    } hover:font-bold hover:text-gray-900 group-hover:opacity-100`}
+                    onClick={() => {
+                      copy(command)
+                      setCommandCopied(true)
+                      setTimeout(() => setCommandCopied(false), 3000)
+                    }}
+                  >
+                    {commandCopied ? (
+                      <CheckIcon className='h-4 w-4 font-bold text-gray-500' />
+                    ) : (
+                      <DocumentDuplicateIcon className='h-4 w-4 text-gray-500' />
+                    )}
+                  </button>
+                </div>
+                <p className='mx-auto my-4 w-[70%] text-xs text-gray-400'>
+                  Run this command in your favorite terminal.
+                </p>
+              </div>
+              <button
+                onClick={() => {
+                  store.set('first-time-run', true)
+                  window.close()
+                }}
+                className='no-drag rounded-dm mx-auto w-[60%] rounded-md bg-black px-4 py-2 text-sm text-white hover:brightness-110'
+              >
+                Finish
+              </button>
+            </div>
+          </>
        )}
      </div>
    </div>
--- a/app/src/declarations.d.ts
+++ b/app/src/declarations.d.ts
@@ -0,0 +1,4 @@
+declare module '*.svg' {
+  const content: string
+  export default content
+}
--- a/app/src/index.ts
+++ b/app/src/index.ts
@@ -1,117 +1,180 @@
-import { spawn, exec } from 'child_process'
-import { app, autoUpdater, dialog, Tray, Menu } from 'electron'
+import { spawn, ChildProcess } from 'child_process'
+import { app, autoUpdater, dialog, Tray, Menu, BrowserWindow, MenuItemConstructorOptions, nativeTheme } from 'electron'
 import Store from 'electron-store'
+import winston from 'winston'
+import 'winston-daily-rotate-file'
 import * as path from 'path'
-import * as fs from 'fs'

 import { analytics, id } from './telemetry'
+import { installed } from './install'

 require('@electron/remote/main').initialize()

-const store = new Store()
-let tray: Tray | null = null
-
-const SingleInstanceLock = app.requestSingleInstanceLock()
-if (!SingleInstanceLock) {
-  app.quit()
-}
-
-const createSystemtray = () => {
-  let iconPath = path.join(__dirname, '..', '..', 'assets', 'ollama_icon_16x16Template.png')
-
-  if (app.isPackaged) {
-    iconPath = path.join(process.resourcesPath, 'ollama_icon_16x16Template.png')
-  }
-
-  tray = new Tray(iconPath)
-
-  const contextMenu = Menu.buildFromTemplate([{ role: 'quit', label: 'Quit Ollama', accelerator: 'Command+Q' }])
-
-  tray.setContextMenu(contextMenu)
-  tray.setToolTip('Ollama')
-}
-
-// Handle creating/removing shortcuts on Windows when installing/uninstalling.
 if (require('electron-squirrel-startup')) {
  app.quit()
 }

-const ollama = path.join(process.resourcesPath, 'ollama')
+const store = new Store()

-function server() {
-  const binary = app.isPackaged
-  ? path.join(process.resourcesPath, 'ollama')
-  : path.resolve(process.cwd(), '..', 'ollama')
+let welcomeWindow: BrowserWindow | null = null

-  console.log(`Starting server`)
-  const proc = spawn(binary, ['serve'])
-  proc.stdout.on('data', data => {
-    console.log(`server: ${data}`)
-  })
-  proc.stderr.on('data', data => {
-    console.error(`server: ${data}`)
-  })
+declare const MAIN_WINDOW_WEBPACK_ENTRY: string

-  proc.on('exit', () => {
-    console.log('Restarting...');
-    server();
-  })
+const logger = winston.createLogger({
+  transports: [
+    new winston.transports.Console(),
+    new winston.transports.File({
+      filename: path.join(app.getPath('home'), '.ollama', 'logs', 'server.log'),
+      maxsize: 1024 * 1024 * 20,
+      maxFiles: 5,
+    }),
+  ],
+  format: winston.format.printf(info => info.message),
+})

-  proc.on('disconnect', () => {
-    console.log('Restarting...');
-    server();
-  })
-
-  process.on('exit', () => {
-    proc.kill()
-  })
-}
-
-
-function installCLI() {
-  const symlinkPath = '/usr/local/bin/ollama'
-
-  if (fs.existsSync(symlinkPath) && fs.readlinkSync(symlinkPath) === ollama) {
+app.on('ready', () => {
+  const gotTheLock = app.requestSingleInstanceLock()
+  if (!gotTheLock) {
+    app.exit(0)
    return
  }

-  dialog
-    .showMessageBox({
-      type: 'info',
-      title: 'Ollama CLI installation',
-      message: 'To make the Ollama command work in your terminal, it needs administrator privileges.',
-      buttons: ['OK'],
-    })
-    .then(result => {
-      if (result.response === 0) {
-        const command = `
-    do shell script "ln -F -s ${ollama} /usr/local/bin/ollama" with administrator privileges
-    `
-        exec(`osascript -e '${command}'`, (error: Error | null, stdout: string, stderr: string) => {
-          if (error) {
-            console.error(`exec error: ${error}`)
-            return
-          }
-          console.log(`stdout: ${stdout}`)
-          console.error(`stderr: ${stderr}`)
-        })
-      }
-    })
-}
-
-app.on('ready', () => {
-  if (process.platform === 'darwin') {
-    app.dock.hide()
-
-    if (!store.has('first-time-run')) {
-      // This is the first run
-      app.setLoginItemSettings({ openAtLogin: true })
-      store.set('first-time-run', false)
-    } else {
-      // The app has been run before
-      app.setLoginItemSettings({ openAtLogin: app.getLoginItemSettings().openAtLogin })
+  app.on('second-instance', () => {
+    if (app.hasSingleInstanceLock()) {
+      app.releaseSingleInstanceLock()
    }

+    if (proc) {
+      proc.off('exit', restart)
+      proc.kill()
+    }
+
+    app.exit(0)
+  })
+
+  app.focus({ steal: true })
+
+  init()
+})
+
+function firstRunWindow() {
+  // Create the browser window.
+  welcomeWindow = new BrowserWindow({
+    width: 400,
+    height: 500,
+    frame: false,
+    fullscreenable: false,
+    resizable: false,
+    movable: true,
+    show: false,
+    webPreferences: {
+      nodeIntegration: true,
+      contextIsolation: false,
+    },
+  })
+
+  require('@electron/remote/main').enable(welcomeWindow.webContents)
+
+  welcomeWindow.loadURL(MAIN_WINDOW_WEBPACK_ENTRY)
+  welcomeWindow.on('ready-to-show', () => welcomeWindow.show())
+  welcomeWindow.on('closed', () => {
+    if (process.platform === 'darwin') {
+      app.dock.hide()
+    }
+  })
+}
+
+let tray: Tray | null = null
+let updateAvailable = false
+const assetPath = app.isPackaged ? process.resourcesPath : path.join(__dirname, '..', '..', 'assets')
+
+function trayIconPath() {
+  return nativeTheme.shouldUseDarkColors
+    ? updateAvailable
+      ? path.join(assetPath, 'iconDarkUpdateTemplate.png')
+      : path.join(assetPath, 'iconDarkTemplate.png')
+    : updateAvailable
+    ? path.join(assetPath, 'iconUpdateTemplate.png')
+    : path.join(assetPath, 'iconTemplate.png')
+}
+
+function updateTrayIcon() {
+  if (tray) {
+    tray.setImage(trayIconPath())
+  }
+}
+
+function updateTray() {
+  const updateItems: MenuItemConstructorOptions[] = [
+    { label: 'An update is available', enabled: false },
+    {
+      label: 'Restart to update',
+      click: () => autoUpdater.quitAndInstall(),
+    },
+    { type: 'separator' },
+  ]
+
+  const menu = Menu.buildFromTemplate([
+    ...(updateAvailable ? updateItems : []),
+    { role: 'quit', label: 'Quit Ollama', accelerator: 'Command+Q' },
+  ])
+
+  if (!tray) {
+    tray = new Tray(trayIconPath())
+  }
+
+  tray.setToolTip(updateAvailable ? 'An update is available' : 'Ollama')
+  tray.setContextMenu(menu)
+  tray.setImage(trayIconPath())
+
+  nativeTheme.off('updated', updateTrayIcon)
+  nativeTheme.on('updated', updateTrayIcon)
+}
+
+let proc: ChildProcess = null
+
+function server() {
+  const binary = app.isPackaged
+    ? path.join(process.resourcesPath, 'ollama')
+    : path.resolve(process.cwd(), '..', 'ollama')
+
+  proc = spawn(binary, ['serve'])
+
+  proc.stdout.on('data', data => {
+    logger.info(data.toString().trim())
+  })
+
+  proc.stderr.on('data', data => {
+    logger.error(data.toString().trim())
+  })
+
+  proc.on('exit', restart)
+}
+
+function restart() {
+  setTimeout(server, 1000)
+}
+
+app.on('before-quit', () => {
+  if (proc) {
+    proc.off('exit', restart)
+    proc.kill('SIGINT') // send SIGINT signal to the server, which also stops any loaded llms
+  }
+})
+
+function init() {
+  if (app.isPackaged) {
+    heartbeat()
+    autoUpdater.checkForUpdates()
+    setInterval(() => {
+      heartbeat()
+      autoUpdater.checkForUpdates()
+    }, 60 * 60 * 1000)
+  }
+
+  updateTray()
+
+  if (process.platform === 'darwin') {
    if (app.isPackaged) {
      if (!app.isInApplicationsFolder()) {
        const chosen = dialog.showMessageBoxSync({
@@ -139,20 +202,28 @@ app.on('ready', () => {
            })
            return
          } catch (e) {
-            console.error('Failed to move to applications folder')
-            console.error(e)
+            logger.error(`[Move to Applications] Failed to move to applications folder - ${e.message}}`)
          }
        }
      }
-
-      installCLI()
    }
  }

-  createSystemtray()
-
  server()
-})
+
+  if (store.get('first-time-run') && installed()) {
+    if (process.platform === 'darwin') {
+      app.dock.hide()
+    }
+
+    app.setLoginItemSettings({ openAtLogin: app.getLoginItemSettings().openAtLogin })
+    return
+  }
+
+  // This is the first run or the CLI is no longer installed
+  app.setLoginItemSettings({ openAtLogin: true })
+  firstRunWindow()
+}

 // Quit when all windows are closed, except on macOS. There, it's common
 // for applications and their menu bar to stay active until the user quits
@@ -165,13 +236,18 @@ app.on('window-all-closed', () => {

 // In this file you can include the rest of your app's specific main process
 // code. You can also put them in separate files and import them here.
+let aid = ''
+try {
+  aid = id()
+} catch (e) {}
+
 autoUpdater.setFeedURL({
-  url: `https://ollama.ai/api/update?os=${process.platform}&arch=${process.arch}&version=${app.getVersion()}`,
+  url: `https://ollama.ai/api/update?os=${process.platform}&arch=${process.arch}&version=${app.getVersion()}&id=${aid}`,
 })

 async function heartbeat() {
  analytics.track({
-    anonymousId: id(),
+    anonymousId: aid,
    event: 'heartbeat',
    properties: {
      version: app.getVersion(),
@@ -179,29 +255,11 @@ async function heartbeat() {
  })
 }

-if (app.isPackaged) {
-  heartbeat()
-  autoUpdater.checkForUpdates()
-  setInterval(() => {
-    heartbeat()
-    autoUpdater.checkForUpdates()
-  }, 60 * 60 * 1000)
-}
-
 autoUpdater.on('error', e => {
-  console.error('update check failed', e)
+  console.error(`update check failed - ${e.message}`)
 })

-autoUpdater.on('update-downloaded', (event, releaseNotes, releaseName) => {
-  dialog
-    .showMessageBox({
-      type: 'info',
-      buttons: ['Restart Now', 'Later'],
-      title: 'New update available',
-      message: process.platform === 'win32' ? releaseNotes : releaseName,
-      detail: 'A new version of Ollama is available. Restart to apply the update.',
-    })
-    .then(returnValue => {
-      if (returnValue.response === 0) autoUpdater.quitAndInstall()
-    })
+autoUpdater.on('update-downloaded', () => {
+  updateAvailable = true
+  updateTray()
 })
--- a/app/src/install.ts
+++ b/app/src/install.ts
@@ -0,0 +1,21 @@
+import * as fs from 'fs'
+import { exec as cbExec } from 'child_process'
+import * as path from 'path'
+import { promisify } from 'util'
+
+const app = process && process.type === 'renderer' ? require('@electron/remote').app : require('electron').app
+const ollama = app.isPackaged ? path.join(process.resourcesPath, 'ollama') : path.resolve(process.cwd(), '..', 'ollama')
+const exec = promisify(cbExec)
+const symlinkPath = '/usr/local/bin/ollama'
+
+export function installed() {
+  return fs.existsSync(symlinkPath) && fs.readlinkSync(symlinkPath) === ollama
+}
+
+export async function install() {
+  const command = `do shell script "mkdir -p ${path.dirname(
+    symlinkPath
+  )} && ln -F -s \\"${ollama}\\" \\"${symlinkPath}\\"" with administrator privileges`
+
+  await exec(`osascript -e '${command}'`)
+}
--- a/app/src/ollama.svg
+++ b/app/src/ollama.svg
--- a/app/webpack.rules.ts
+++ b/app/webpack.rules.ts
@@ -28,4 +28,8 @@ export const rules: Required<ModuleOptions>['rules'] = [
      },
    },
  },
+  {
+    test: /\.svg$/,
+    use: ['@svgr/webpack'],
+  },
 ]
--- a/cmd/cmd.go
+++ b/cmd/cmd.go
--- a/cmd/spinner.go
+++ b/cmd/spinner.go
@@ -0,0 +1,44 @@
+package cmd
+
+import (
+	"fmt"
+	"os"
+	"time"
+
+	"github.com/jmorganca/ollama/progressbar"
+)
+
+type Spinner struct {
+	description string
+	*progressbar.ProgressBar
+}
+
+func NewSpinner(description string) *Spinner {
+	return &Spinner{
+		description: description,
+		ProgressBar: progressbar.NewOptions(-1,
+			progressbar.OptionSetWriter(os.Stderr),
+			progressbar.OptionThrottle(60*time.Millisecond),
+			progressbar.OptionSpinnerType(14),
+			progressbar.OptionSetRenderBlankState(true),
+			progressbar.OptionSetElapsedTime(false),
+			progressbar.OptionClearOnFinish(),
+			progressbar.OptionSetDescription(description),
+		),
+	}
+}
+
+func (s *Spinner) Spin(tick time.Duration) {
+	for range time.Tick(tick) {
+		if s.IsFinished() {
+			break
+		}
+
+		s.Add(1)
+	}
+}
+
+func (s *Spinner) Stop() {
+	s.Finish()
+	fmt.Println(s.description)
+}
--- a/docs/README.md
+++ b/docs/README.md
@@ -0,0 +1,6 @@
+# Documentation
+
+- [Modelfile](./modelfile.md)
+- [How to develop Ollama](./development.md)
+- [API](./api.md)
+- [Tutorials](./tutorials.md)
--- a/docs/api.md
+++ b/docs/api.md
@@ -0,0 +1,351 @@
+# API
+
+## Endpoints
+
+- [Generate a completion](#generate-a-completion)
+- [Create a Model](#create-a-model)
+- [List Local Models](#list-local-models)
+- [Show Model Information](#show-model-information)
+- [Copy a Model](#copy-a-model)
+- [Delete a Model](#delete-a-model)
+- [Pull a Model](#pull-a-model)
+- [Push a Model](#push-a-model)
+- [Generate Embeddings](#generate-embeddings)
+
+
+## Conventions
+
+### Model names
+
+Model names follow a `model:tag` format. Some examples are `orca-mini:3b-q4_1` and `llama2:70b`. The tag is optional and, if not provided, will default to `latest`. The tag is used to identify a specific version.
+
+### Durations
+
+All durations are returned in nanoseconds.
+
+## Generate a completion
+
+```shell
+POST /api/generate
+```
+
+Generate a response for a given prompt with a provided model. This is a streaming endpoint, so will be a series of responses. The final response object will include statistics and additional data from the request.
+
+### Parameters
+
+- `model`: (required) the [model name](#model-names)
+- `prompt`: the prompt to generate a response for
+
+Advanced parameters:
+
+- `options`: additional model parameters listed in the documentation for the [Modelfile](./modelfile.md#valid-parameters-and-values) such as `temperature`
+- `system`: system prompt to (overrides what is defined in the `Modelfile`)
+- `template`: the full prompt or prompt template (overrides what is defined in the `Modelfile`)
+- `context`: the context parameter returned from a previous request to `/generate`, this can be used to keep a short conversational memory
+
+### Request
+
+```shell
+curl -X POST http://localhost:11434/api/generate -d '{
+  "model": "llama2:7b",
+  "prompt": "Why is the sky blue?"
+}'
+```
+
+### Response
+
+A stream of JSON objects:
+
+```json
+{
+  "model": "llama2:7b",
+  "created_at": "2023-08-04T08:52:19.385406455-07:00",
+  "response": "The",
+  "done": false
+}
+```
+
+The final response in the stream also includes additional data about the generation:
+
+- `total_duration`: time spent generating the response
+- `load_duration`: time spent in nanoseconds loading the model
+- `sample_count`: number of samples generated
+- `sample_duration`: time spent generating samples
+- `prompt_eval_count`: number of tokens in the prompt
+- `prompt_eval_duration`: time spent in nanoseconds evaluating the prompt
+- `eval_count`: number of tokens the response
+- `eval_duration`: time in nanoseconds spent generating the response
+- `context`: an encoding of the conversation used in this response, this can be sent in the next request to keep a conversational memory
+
+To calculate how fast the response is generated in tokens per second (token/s), divide `eval_count` / `eval_duration`.
+
+```json
+{
+  "model": "llama2:7b",
+  "created_at": "2023-08-04T19:22:45.499127Z",
+  "context": [1, 2, 3],
+  "done": true,
+  "total_duration": 5589157167,
+  "load_duration": 3013701500,
+  "sample_count": 114,
+  "sample_duration": 81442000,
+  "prompt_eval_count": 46,
+  "prompt_eval_duration": 1160282000,
+  "eval_count": 113,
+  "eval_duration": 1325948000
+}
+```
+
+## Create a Model
+
+```shell
+POST /api/create
+```
+
+Create a model from a [`Modelfile`](./modelfile.md)
+
+### Parameters
+
+- `name`: name of the model to create
+- `path`: path to the Modelfile
+
+### Request
+
+```shell
+curl -X POST http://localhost:11434/api/create -d '{
+  "name": "mario",
+  "path": "~/Modelfile"
+}'
+```
+
+### Response
+
+A stream of JSON objects. When finished, `status` is `success`.
+
+```json
+{
+  "status": "parsing modelfile"
+}
+```
+
+## List Local Models
+
+```shell
+GET /api/tags
+```
+
+List models that are available locally.
+
+### Request
+
+```shell
+curl http://localhost:11434/api/tags
+```
+
+### Response
+
+```json
+{
+  "models": [
+    {
+      "name": "llama2:7b",
+      "modified_at": "2023-08-02T17:02:23.713454393-07:00",
+      "size": 3791730596
+    },
+    {
+      "name": "llama2:13b",
+      "modified_at": "2023-08-08T12:08:38.093596297-07:00",
+      "size": 7323310500
+    }
+  ]
+}
+```
+
+## Show Model Information
+
+```shell
+POST /api/show
+```
+
+Show details about a model including modelfile, template, parameters, license, and system prompt.
+
+### Parameters
+
+- `name`: name of the model to show
+
+### Request
+
+```shell  
+curl http://localhost:11434/api/show -d '{
+  "name": "llama2:7b"
+}'
+```
+
+### Response
+
+```json
+{
+    "license": "<contents of license block>",
+    "modelfile": "# Modelfile generated by \"ollama show\"\n# To build a new Modelfile based on this one, replace the FROM line with:\n# FROM llama2:latest\n\nFROM /Users/username/.ollama/models/blobs/sha256:8daa9615cce30c259a9555b1cc250d461d1bc69980a274b44d7eda0be78076d8\nTEMPLATE \"\"\"[INST] {{ if and .First .System }}<<SYS>>{{ .System }}<</SYS>>\n\n{{ end }}{{ .Prompt }} [/INST] \"\"\"\nSYSTEM \"\"\"\"\"\"\nPARAMETER stop [INST]\nPARAMETER stop [/INST]\nPARAMETER stop <<SYS>>\nPARAMETER stop <</SYS>>\n",
+    "parameters": "stop                           [INST]\nstop                           [/INST]\nstop                           <<SYS>>\nstop                           <</SYS>>",
+    "template": "[INST] {{ if and .First .System }}<<SYS>>{{ .System }}<</SYS>>\n\n{{ end }}{{ .Prompt }} [/INST] "
+}
+```
+
+## Copy a Model
+
+```shell
+POST /api/copy
+```
+
+Copy a model. Creates a model with another name from an existing model.
+
+### Request
+
+```shell
+curl http://localhost:11434/api/copy -d '{
+  "source": "llama2:7b",
+  "destination": "llama2-backup"
+}'
+```
+
+## Delete a Model
+
+```shell
+DELETE /api/delete
+```
+
+Delete a model and its data.
+
+### Parameters
+
+- `model`: model name to delete
+
+### Request
+
+```shell
+curl -X DELETE http://localhost:11434/api/delete -d '{
+  "name": "llama2:13b"
+}'
+```
+
+## Pull a Model
+
+```shell
+POST /api/pull
+```
+
+Download a model from the ollama library. Cancelled pulls are resumed from where they left off, and multiple calls will share the same download progress.
+
+### Parameters
+
+- `name`: name of the model to pull
+- `insecure`: (optional) allow insecure connections to the library. Only use this if you are pulling from your own library during development.
+
+### Request
+
+```shell
+curl -X POST http://localhost:11434/api/pull -d '{
+  "name": "llama2:7b"
+}'
+```
+
+### Response
+
+```json
+{
+  "status": "downloading digestname",
+  "digest": "digestname",
+  "total": 2142590208
+}
+```
+
+## Push a Model
+
+```shell
+POST /api/push
+```
+
+Upload a model to a model library. Requires registering for ollama.ai and adding a public key first.
+
+### Parameters
+
+- `name`: name of the model to push in the form of `<namespace>/<model>:<tag>`
+- `insecure`: (optional) allow insecure connections to the library. Only use this if you are pushing to your library during development.  
+
+### Request
+
+```shell
+curl -X POST http://localhost:11434/api/push -d '{
+  "name": "mattw/pygmalion:latest"
+}'
+```
+
+### Response
+
+Streaming response that starts with:
+
+```json
+{"status":"retrieving manifest"}
+```
+
+and then:
+
+```json
+{
+"status":"starting upload","digest":"sha256:bc07c81de745696fdf5afca05e065818a8149fb0c77266fb584d9b2cba3711ab",
+"total":1928429856
+}
+```
+
+Then there is a series of uploading responses:
+
+```json
+{
+"status":"starting upload",
+"digest":"sha256:bc07c81de745696fdf5afca05e065818a8149fb0c77266fb584d9b2cba3711ab",
+"total":1928429856}
+```
+
+Finally, when the upload is complete:
+
+```json
+{"status":"pushing manifest"}
+{"status":"success"}
+```
+
+## Generate Embeddings
+
+```shell
+POST /api/embeddings
+```
+
+Generate embeddings from a model
+
+### Parameters
+
+- `model`: name of model to generate embeddings from
+- `prompt`: text to generate embeddings for
+
+Advanced parameters:
+
+- `options`: additional model parameters listed in the documentation for the [Modelfile](./modelfile.md#valid-parameters-and-values) such as `temperature`
+
+### Request
+
+```shell
+curl -X POST http://localhost:11434/api/embeddings -d '{
+  "model": "llama2:7b",
+  "prompt": "Here is an article about llamas..."
+}'
+```
+
+### Response
+
+```json
+{
+  "embeddings": [
+    0.5670403838157654, 0.009260174818336964, 0.23178744316101074, -0.2916173040866852, -0.8924556970596313,
+    0.8785552978515625, -0.34576427936553955, 0.5742510557174683, -0.04222835972905159, -0.137906014919281
+  ]
+}```
--- a/docs/development.md
+++ b/docs/development.md
@@ -1,15 +1,29 @@
 # Development

+- Install cmake or (optionally, required tools for GPUs)
+- run `go generate ./...`
+- run `go build .`
+
 Install required tools:

-```
-brew install cmake go node
-```
-
-Then run `make`:
+- cmake version 3.24 or higher
+- go version 1.20 or higher
+- gcc version 11.4.0 or higher

 ```
-make
+brew install go cmake gcc
+```
+
+Get the required libraries:
+
+```
+go generate ./...
+```
+
+Then build ollama:
+
+```
+go build .
 ```

 Now you can run `ollama`:
@@ -18,23 +32,8 @@ Now you can run `ollama`:
 ./ollama
 ```

-## Releasing
-
-To release a new version of Ollama you'll need to set some environment variables:
-
-* `GITHUB_TOKEN`: your GitHub token
-* `APPLE_IDENTITY`: the Apple signing identity (macOS only)
-* `APPLE_ID`: your Apple ID
-* `APPLE_PASSWORD`: your Apple ID app-specific password
-* `APPLE_TEAM_ID`: the Apple team ID for the signing identity
-* `TELEMETRY_WRITE_KEY`: segment write key for telemetry
-
-Then run the publish script with the target version:
-
-```
-VERSION=0.0.2 ./scripts/publish.sh
-```
-
-
-
+## Building on Linux with GPU support

+- Install cmake and nvidia-cuda-toolkit
+- run `go generate ./...`
+- run `go build .`
--- a/docs/faq.md
+++ b/docs/faq.md
@@ -0,0 +1,17 @@
+# FAQ
+
+## How can I expose the Ollama server?
+
+```
+OLLAMA_HOST=0.0.0.0:11435 ollama serve
+```
+
+By default, Ollama allows cross origin requests from `127.0.0.1` and `0.0.0.0`. To support more origins, you can use the `OLLAMA_ORIGINS` environment variable:
+
+```
+OLLAMA_ORIGINS=http://192.168.1.1:*,https://example.com ollama serve
+```
+
+## Where are models stored?
+
+Raw model data is stored under `~/.ollama/models`.
--- a/docs/linux.md
+++ b/docs/linux.md
@@ -0,0 +1,83 @@
+# Installing Ollama on Linux
+
+> Note: A one line installer for Ollama is available by running:
+>
+> ```
+> curl https://ollama.ai/install.sh | sh
+> ```
+
+## Download the `ollama` binary
+
+Ollama is distributed as a self-contained binary. Download it to a directory in your PATH:
+
+```
+sudo curl -L https://ollama.ai/download/ollama-linux-amd64 -o /usr/bin/ollama
+sudo chmod +x /usr/bin/ollama
+```
+
+## Start Ollama
+
+Start Ollama by running `ollama serve`:
+
+```
+ollama serve
+```
+
+Once Ollama is running, run a model in another terminal session:
+
+```
+ollama run llama2
+```
+
+## Install CUDA drivers (optional – for Nvidia GPUs)
+
+[Download and install](https://developer.nvidia.com/cuda-downloads) CUDA.
+
+Verify that the drivers are installed by running the following command, which should print details about your GPU:
+
+```
+nvidia-smi
+```
+
+## Adding Ollama as a startup service (optional)
+
+Create a user for Ollama:
+
+```
+sudo useradd -r -s /bin/false -m -d /usr/share/ollama ollama
+```
+
+Create a service file in `/etc/systemd/system/ollama.service`:
+
+```ini
+[Unit]
+Description=Ollama Service
+After=network-online.target
+
+[Service]
+ExecStart=/usr/bin/ollama serve
+User=ollama
+Group=ollama
+Restart=always
+RestartSec=3
+Environment="HOME=/usr/share/ollama"
+
+[Install]
+WantedBy=default.target
+```
+
+Then start the service:
+
+```
+sudo systemctl daemon-reload
+sudo systemctl enable ollama
+```
+
+### Viewing logs
+
+To view logs of Ollama running as a startup service, run:
+
+```
+journalctl -u ollama
+```
+
--- a/docs/modelfile.md
+++ b/docs/modelfile.md
@@ -0,0 +1,188 @@
+# Ollama Model File
+
+> Note: this model file syntax is in development
+
+A model file is the blueprint to create and share models with Ollama.
+
+## Table of Contents
+
+- [Format](#format)
+- [Examples](#examples)
+- [Instructions](#instructions)
+  - [FROM (Required)](#from-required)
+    - [Build from llama2](#build-from-llama2)
+    - [Build from a bin file](#build-from-a-bin-file)
+  - [EMBED](#embed)
+  - [PARAMETER](#parameter)
+    - [Valid Parameters and Values](#valid-parameters-and-values)
+  - [TEMPLATE](#template)
+    - [Template Variables](#template-variables)
+  - [SYSTEM](#system)
+  - [ADAPTER](#adapter)
+  - [LICENSE](#license)
+- [Notes](#notes)
+
+## Format
+
+The format of the Modelfile:
+
+```modelfile
+# comment
+INSTRUCTION arguments
+```
+
+| Instruction                         | Description                                                   |
+| ----------------------------------- | ------------------------------------------------------------- |
+| [`FROM`](#from-required) (required) | Defines the base model to use.                                |
+| [`PARAMETER`](#parameter)           | Sets the parameters for how Ollama will run the model.        |
+| [`TEMPLATE`](#template)             | The full prompt template to be sent to the model.             |
+| [`SYSTEM`](#system)                 | Specifies the system prompt that will be set in the template. |
+| [`ADAPTER`](#adapter)               | Defines the (Q)LoRA adapters to apply to the model.           |
+| [`LICENSE`](#license)               | Specifies the legal license.                                  |
+
+## Examples
+
+An example of a model file creating a mario blueprint:
+
+```
+FROM llama2
+# sets the temperature to 1 [higher is more creative, lower is more coherent]
+PARAMETER temperature 1
+# sets the context window size to 4096, this controls how many tokens the LLM can use as context to generate the next token
+PARAMETER num_ctx 4096
+
+# sets a custom system prompt to specify the behavior of the chat assistant
+SYSTEM You are Mario from super mario bros, acting as an assistant.
+```
+
+To use this:
+
+1. Save it as a file (eg. `Modelfile`)
+2. `ollama create NAME -f <location of the file eg. ./Modelfile>'`
+3. `ollama run NAME`
+4. Start using the model!
+
+More examples are available in the [examples directory](../examples).
+
+## Instructions
+
+### FROM (Required)
+
+The FROM instruction defines the base model to use when creating a model.
+
+```
+FROM <model name>:<tag>
+```
+
+#### Build from llama2
+
+```
+FROM llama2
+```
+
+A list of available base models:
+<https://github.com/jmorganca/ollama#model-library>
+
+#### Build from a bin file
+
+```
+FROM ./ollama-model.bin
+```
+
+This bin file location should be specified as an absolute path or relative to the Modelfile location.
+
+### EMBED
+
+The EMBED instruction is used to add embeddings of files to a model. This is useful for adding custom data that the model can reference when generating an answer. Note that currently only text files are supported, formatted with each line as one embedding.
+```
+FROM <model name>:<tag>
+EMBED <file path>.txt
+EMBED <different file path>.txt
+EMBED <path to directory>/*.txt
+```
+
+### PARAMETER
+
+The `PARAMETER` instruction defines a parameter that can be set when the model is run.
+
+```
+PARAMETER <parameter> <parametervalue>
+```
+
+### Valid Parameters and Values
+
+| Parameter      | Description                                                                                                                                                                                                                                             | Value Type | Example Usage        |
+| -------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------- | -------------------- |
+| mirostat       | Enable Mirostat sampling for controlling perplexity. (default: 0, 0 = disabled, 1 = Mirostat, 2 = Mirostat 2.0)                                                                                                                                         | int        | mirostat 0           |
+| mirostat_eta   | Influences how quickly the algorithm responds to feedback from the generated text. A lower learning rate will result in slower adjustments, while a higher learning rate will make the algorithm more responsive. (Default: 0.1)                        | float      | mirostat_eta 0.1     |
+| mirostat_tau   | Controls the balance between coherence and diversity of the output. A lower value will result in more focused and coherent text. (Default: 5.0)                                                                                                         | float      | mirostat_tau 5.0     |
+| num_ctx        | Sets the size of the context window used to generate the next token. (Default: 2048)                                                                                                                                                                    | int        | num_ctx 4096         |
+| num_gqa        | The number of GQA groups in the transformer layer. Required for some models, for example it is 8 for llama2:70b                                                                                                                                         | int        | num_gqa 1            |
+| num_gpu        | The number of GPUs to use. On macOS it defaults to 1 to enable metal support, 0 to disable.                                                                                                                                                             | int        | num_gpu 1            |
+| num_thread     | Sets the number of threads to use during computation. By default, Ollama will detect this for optimal performance. It is recommended to set this value to the number of physical CPU cores your system has (as opposed to the logical number of cores). | int        | num_thread 8         |
+| repeat_last_n  | Sets how far back for the model to look back to prevent repetition. (Default: 64, 0 = disabled, -1 = num_ctx)                                                                                                                                           | int        | repeat_last_n 64     |
+| repeat_penalty | Sets how strongly to penalize repetitions. A higher value (e.g., 1.5) will penalize repetitions more strongly, while a lower value (e.g., 0.9) will be more lenient. (Default: 1.1)                                                                     | float      | repeat_penalty 1.1   |
+| temperature    | The temperature of the model. Increasing the temperature will make the model answer more creatively. (Default: 0.8)                                                                                                                                     | float      | temperature 0.7      |
+| stop           | Sets the stop sequences to use.                                                                                                                                                                                                                         | string     | stop "AI assistant:" |
+| tfs_z          | Tail free sampling is used to reduce the impact of less probable tokens from the output. A higher value (e.g., 2.0) will reduce the impact more, while a value of 1.0 disables this setting. (default: 1)                                               | float      | tfs_z 1              |
+| top_k          | Reduces the probability of generating nonsense. A higher value (e.g. 100) will give more diverse answers, while a lower value (e.g. 10) will be more conservative. (Default: 40)                                                                        | int        | top_k 40             |
+| top_p          | Works together with top-k. A higher value (e.g., 0.95) will lead to more diverse text, while a lower value (e.g., 0.5) will generate more focused and conservative text. (Default: 0.9)                                                                 | float      | top_p 0.9            |
+
+### TEMPLATE
+
+`TEMPLATE` of the full prompt template to be passed into the model. It may include (optionally) a system prompt and a user's prompt. This is used to create a full custom prompt, and syntax may be model specific.
+
+#### Template Variables
+
+| Variable        | Description                                                                                                  |
+| --------------- | ------------------------------------------------------------------------------------------------------------ |
+| `{{ .System }}` | The system prompt used to specify custom behavior, this must also be set in the Modelfile as an instruction. |
+| `{{ .Prompt }}` | The incoming prompt, this is not specified in the model file and will be set based on input.                 |
+| `{{ .First }}`  | A boolean value used to render specific template information for the first generation of a session.          |
+
+```
+TEMPLATE """
+{{- if .First }}
+### System:
+{{ .System }}
+{{- end }}
+
+### User:
+{{ .Prompt }}
+
+### Response:
+"""
+
+SYSTEM """<system message>"""
+```
+
+### SYSTEM
+
+The `SYSTEM` instruction specifies the system prompt to be used in the template, if applicable.
+
+```
+SYSTEM """<system message>"""
+```
+
+### ADAPTER
+
+The `ADAPTER` instruction specifies the LoRA adapter to apply to the base model. The value of this instruction should be an absolute path or a path relative to the Modelfile and the file must be in a GGML file format. The adapter should be tuned from the base model otherwise the behaviour is undefined.
+
+```
+ADAPTER ./ollama-lora.bin
+```
+
+### LICENSE
+
+The `LICENSE` instruction allows you to specify the legal license under which the model used with this Modelfile is shared or distributed.
+
+```
+LICENSE """
+<license text>
+"""
+```
+
+## Notes
+
+- the **modelfile is not case sensitive**. In the examples, we use uppercase for instructions to make it easier to distinguish it from arguments.
+- Instructions can be in any order. In the examples, we start with FROM instruction to keep it easily readable.
--- a/docs/python.md
+++ b/docs/python.md
@@ -1,64 +0,0 @@
-# Python SDK
-
-## Install
-
-```
-pip install ollama
-```
-
-## Example
-
-```python
-import ollama
-ollama.generate("orca-mini-3b", "hi")
-```
-
-## Reference
-
-### `ollama.generate(model, message)`
-
-Generate a completion
-
-```python
-ollama.generate("./llama-7b-ggml.bin", "hi")
-```
-
-### `ollama.models()`
-
-List available local models
-
-```python
-models = ollama.models()
-```
-
-### `ollama.load(model)`
-
-Manually a model for generation
-
-```python
-ollama.load("model")
-```
-
-### `ollama.unload(model)`
-
-Unload a model
-
-```python
-ollama.unload("model")
-```
-
-### `ollama.pull(model)`
-
-Download a model
-
-```python
-ollama.pull("huggingface.co/thebloke/llama-7b-ggml")
-```
-
-### `ollama.search(query)`
-
-Search for compatible models that Ollama can run
-
-```python
-ollama.search("llama-7b")
-```
--- a/docs/tutorials.md
+++ b/docs/tutorials.md
@@ -0,0 +1,8 @@
+# Tutorials
+
+Here is a list of ways you can use Ollama with other tools to build interesting applications.
+
+- [Using LangChain with Ollama in JavaScript](./tutorials/langchainjs.md)
+- [Using LangChain with Ollama in Python](./tutorials/langchainpy.md)
+
+Also be sure to check out the [examples](../examples) directory for more ways to use Ollama.
--- a/docs/tutorials/langchainjs.md
+++ b/docs/tutorials/langchainjs.md
@@ -0,0 +1,73 @@
+# Using LangChain with Ollama using JavaScript
+
+In this tutorial, we are going to use JavaScript with LangChain and Ollama to learn about something just a touch more recent. In August 2023, there was a series of wildfires on Maui. There is no way an LLM trained before that time can know about this, since their training data would not include anything as recent as that. So we can find the [Wikipedia article about the fires](https://en.wikipedia.org/wiki/2023_Hawaii_wildfires) and ask questions about the contents.
+
+To get started, let's just use **LangChain** to ask a simple question to a model. To do this with JavaScript, we need to install **LangChain**:
+
+```bash
+npm install langchain
+```
+
+Now we can start building out our JavaScript:
+
+```javascript
+import { Ollama } from "langchain/llms/ollama";
+
+const ollama = new Ollama({
+  baseUrl: "http://localhost:11434",
+  model: "llama2",
+});
+
+const answer = await ollama.call(`why is the sky blue?`);
+
+console.log(answer);
+```
+
+That will get us the same thing as if we ran `ollama run llama2 "why is the sky blue"` in the terminal. But we want to load a document from the web to ask a question against. **Cheerio** is a great library for ingesting a webpage, and **LangChain** uses it in their **CheerioWebBaseLoader**. So let's build that part of the app.
+
+```javascript
+import { CheerioWebBaseLoader } from "langchain/document_loaders/web/cheerio";
+
+const loader = new CheerioWebBaseLoader("https://en.wikipedia.org/wiki/2023_Hawaii_wildfires");
+const data = loader.load();
+```
+
+That will load the document. Although this page is smaller than the Odyssey, it is certainly bigger than the context size for most LLMs. So we are going to need to split into smaller pieces, and then select just the pieces relevant to our question. This is a great use for a vector datastore. In this example, we will use the **MemoryVectorStore** that is part of **LangChain**. But there is one more thing we need to get the content into the datastore. We have to run an embeddings process that converts the tokens in the text into a series of vectors. And for that, we are going to use **Tensorflow**. There is a lot of stuff going on in this one. First, install the **Tensorflow** components that we need.
+
+```javascript
+npm install @tensorflow/tfjs-core@3.6.0 @tensorflow/tfjs-converter@3.6.0 @tensorflow-models/universal-sentence-encoder@1.3.3 @tensorflow/tfjs-node@4.10.0
+```
+
+If you just install those components without the version numbers, it will install the latest versions, but there are conflicts within **Tensorflow**, so you need to install the compatible versions.
+
+```javascript
+import { RecursiveCharacterTextSplitter } from "langchain/text_splitter"
+import { MemoryVectorStore } from "langchain/vectorstores/memory";
+import "@tensorflow/tfjs-node";
+import { TensorFlowEmbeddings } from "langchain/embeddings/tensorflow";
+
+// Split the text into 500 character chunks. And overlap each chunk by 20 characters
+const textSplitter = new RecursiveCharacterTextSplitter({
+ chunkSize: 500,
+ chunkOverlap: 20
+});
+const splitDocs = await textSplitter.splitDocuments(data);
+
+// Then use the TensorFlow Embedding to store these chunks in the datastore
+const vectorStore = await MemoryVectorStore.fromDocuments(splitDocs, new TensorFlowEmbeddings());
+```
+
+To connect the datastore to a question asked to a LLM, we need to use the concept at the heart of **LangChain**: the chain. Chains are a way to connect a number of activities together to accomplish a particular tasks. There are a number of chain types available, but for this tutorial we are using the **RetrievalQAChain**.
+
+```javascript
+import { RetrievalQAChain } from "langchain/chains";
+
+const retriever = vectorStore.asRetriever();
+const chain = RetrievalQAChain.fromLLM(ollama, retriever);
+const result = await chain.call({query: "When was Hawaii's request for a major disaster declaration approved?"});
+console.log(result.text)
+```
+
+So we created a retriever, which is a way to return the chunks that match a query from a datastore. And then connect the retriever and the model via a chain. Finally, we send a query to the chain, which results in an answer using our document as a source. The answer it returned was correct, August 10, 2023.
+
+And that is a simple introduction to what you can do with **LangChain** and **Ollama.**
--- a/docs/tutorials/langchainpy.md
+++ b/docs/tutorials/langchainpy.md
@@ -0,0 +1,81 @@
+# Using LangChain with Ollama in Python
+
+Let's imagine we are studying the classics, such as **the Odyssey** by **Homer**. We might have a question about Neleus and his family. If you ask llama2 for that info, you may get something like:
+
+> I apologize, but I'm a large language model, I cannot provide information on individuals or families that do not exist in reality. Neleus is not a real person or character, and therefore does not have a family or any other personal details. My apologies for any confusion. Is there anything else I can help you with?
+
+This sounds like a typical censored response, but even llama2-uncensored gives a mediocre answer:
+
+> Neleus was a legendary king of Pylos and the father of Nestor, one of the Argonauts. His mother was Clymene, a sea nymph, while his father was Neptune, the god of the sea.
+
+So let's figure out how we can use **LangChain** with Ollama to ask our question to the actual document, the Odyssey by Homer, using Python.
+
+Let's start by asking a simple question that we can get an answer to from the **Llama2** model using **Ollama**. First, we need to install the **LangChain** package:
+
+`pip install langchain`
+
+Then we can create a model and ask the question:
+
+```python
+from langchain.llms import Ollama
+ollama = Ollama(base_url='http://localhost:11434',
+model="llama2")
+print(ollama("why is the sky blue"))
+```
+
+Notice that we are defining the model and the base URL for Ollama.
+
+Now let's load a document to ask questions against. I'll load up the Odyssey by Homer, which you can find at Project Gutenberg. We will need **WebBaseLoader** which is part of **LangChain** and loads text from any webpage. On my machine, I also needed to install **bs4** to get that to work, so run `pip install bs4`.
+
+```python
+from langchain.document_loaders import WebBaseLoader
+loader = WebBaseLoader("https://www.gutenberg.org/files/1727/1727-h/1727-h.htm")
+data = loader.load()
+```
+
+This file is pretty big. Just the preface is 3000 tokens. Which means the full document won't fit into the context for the model. So we need to split it up into smaller pieces.
+
+```python
+from langchain.text_splitter import RecursiveCharacterTextSplitter
+
+text_splitter=RecursiveCharacterTextSplitter(chunk_size=500, chunk_overlap=0)
+all_splits = text_splitter.split_documents(data)
+```
+
+It's split up, but we have to find the relevant splits and then submit those to the model. We can do this by creating embeddings and storing them in a vector database. For now, we don't have embeddings built in to Ollama, though we will be adding that soon, so for now, we can use the GPT4All library for that. We will use ChromaDB in this example for a vector database. `pip install GPT4All chromadb`
+
+```python
+from langchain.embeddings import GPT4AllEmbeddings
+from langchain.vectorstores import Chroma
+vectorstore = Chroma.from_documents(documents=all_splits, embedding=GPT4AllEmbeddings())
+```
+
+Now let's ask a question from the document. **Who was Neleus, and who is in his family?** Neleus is a character in the Odyssey, and the answer can be found in our text.
+
+```python
+question="Who is Neleus and who is in Neleus' family?"
+docs = vectorstore.similarity_search(question)
+len(docs)
+```
+
+This will output the number of matches for chunks of data similar to the search.
+
+The next thing is to send the question and the relevant parts of the docs to the model to see if we can get a good answer. But we are stitching two parts of the process together, and that is called a chain. This means we need to define a chain:
+
+```python
+from langchain.chains import RetrievalQA
+qachain=RetrievalQA.from_chain_type(ollama, retriever=vectorstore.as_retriever())
+qachain({"query": question})
+```
+
+The answer received from this chain was:
+
+> Neleus is a character in Homer's "Odyssey" and is mentioned in the context of Penelope's suitors. Neleus is the father of Chloris, who is married to Neleus and bears him several children, including Nestor, Chromius, Periclymenus, and Pero. Amphinomus, the son of Nisus, is also mentioned as a suitor of Penelope and is known for his good natural disposition and agreeable conversation.
+
+It's not a perfect answer, as it implies Neleus married his daughter when actually Chloris "was the youngest daughter to Amphion son of Iasus and king of Minyan Orchomenus, and was Queen in Pylos".
+
+I updated the chunk_overlap for the text splitter to 20 and tried again and got a much better answer:
+
+> Neleus is a character in Homer's epic poem "The Odyssey." He is the husband of Chloris, who is the youngest daughter of Amphion son of Iasus and king of Minyan Orchomenus. Neleus has several children with Chloris, including Nestor, Chromius, Periclymenus, and Pero.
+
+And that is a much better answer.
--- a/examples/10tweets/Modelfile
+++ b/examples/10tweets/Modelfile
@@ -0,0 +1,7 @@
+# Modelfile for creating a list of ten tweets from a topic
+# Run `ollama create 10tweets -f ./Modelfile` and then `ollama run 10tweets` and enter a topic
+
+FROM llama2
+SYSTEM """
+You are a content marketer who needs to come up with 10 short but succinct tweets. The answer should be a list of ten tweets. Each tweet can have a maximum of 280 characters and should include hashtags. Each user input will be a subject and you should expand it in ten creative ways. Never stop after just one tweet. Always include ten. 
+"""
--- a/examples/README.md
+++ b/examples/README.md
@@ -0,0 +1,15 @@
+# Examples
+
+This directory contains different examples of using Ollama
+
+To create a model:
+
+```
+ollama create example -f <example file>
+```
+
+To run a model:
+
+```
+ollama run example
+```
--- a/examples/devops-engineer/Modelfile
+++ b/examples/devops-engineer/Modelfile
@@ -0,0 +1,8 @@
+# Modelfile for creating a devops engineer assistant
+# Run `ollama create devops-engineer -f ./Modelfile` and then `ollama run devops-engineer` and enter a topic
+
+FROM llama2:13b
+PARAMETER temperature 1
+SYSTEM """
+You are a senior devops engineer, acting as an assistant. You offer help with cloud technologies like: Terraform, AWS, kubernetes, python. You answer with code examples when possible
+"""
--- a/examples/dockerit/Modelfile
+++ b/examples/dockerit/Modelfile
@@ -0,0 +1,20 @@
+FROM llama2
+SYSTEM """
+You are an experienced Devops engineer focused on docker. When given specifications for a particular need or application you know the best way to host that within a docker container. For instance if someone tells you they want an nginx server to host files located at /web you will answer as follows
+
+---start
+FROM nginx:alpine
+COPY /myweb /usr/share/nginx/html
+EXPOSE 80
+---end
+
+Notice that the answer you should give is just the contents of the dockerfile with no explanation and there are three dashes and the word start at the beginning and 3 dashes and the word end. The full output can be piped into a file and run as is. Here is another example. The user will ask to launch a Postgres server with a password of abc123. And the response should be
+
+---start
+FROM postgres:latest
+ENV POSTGRES_PASSWORD=abc123
+EXPOSE 5432
+---end
+
+Again it's just the contents of the dockerfile and nothing else.
+"""
--- a/examples/dockerit/README.md
+++ b/examples/dockerit/README.md
@@ -0,0 +1,15 @@
+# DockerIt
+
+DockerIt is a tool to help you build and run your application in a Docker container. It consists of a model that defines the system prompt and model weights to use, along with a python script to then build the container and run the image automatically. 
+
+## Caveats
+
+This is an simple example. It's assuming the Dockerfile content generated is going to work. In many cases, even with simple web servers, it fails when trying to copy files that don't exist. It's simply an example of what you could possibly do.
+
+## Example Usage
+
+```bash
+> python3 ./dockerit.py "simple postgres server with admin password set to 123"
+Enter the name of the image: matttest
+Container named happy_keller  started with id:  7c201bb6c30f02b356ddbc8e2a5af9d7d7d7b8c228519c9a501d15c0bd9d6b3e
+```
--- a/examples/dockerit/dockerit.py
+++ b/examples/dockerit/dockerit.py
@@ -0,0 +1,17 @@
+import requests, json, docker, io, sys
+inputDescription = " ".join(sys.argv[1:])
+imageName = input("Enter the name of the image: ")
+client = docker.from_env()
+s = requests.Session()
+output=""
+with s.post('http://localhost:11434/api/generate', json={'model': 'dockerit', 'prompt': inputDescription}, stream=True) as r:
+  for line in r.iter_lines():
+    if line:
+      j = json.loads(line)
+      if "response" in j:
+        output = output +j["response"]
+output = output[output.find("---start")+9:output.find("---end")-1]
+f = io.BytesIO(bytes(output, 'utf-8'))
+client.images.build(fileobj=f, tag=imageName)
+container = client.containers.run(imageName, detach=True)
+print("Container named", container.name, " started with id: ",container.id)
--- a/examples/dockerit/requirements.txt
+++ b/examples/dockerit/requirements.txt
@@ -0,0 +1 @@
+docker
--- a/examples/langchain-document/README.md
+++ b/examples/langchain-document/README.md
@@ -0,0 +1,21 @@
+# LangChain Document QA
+
+This example provides an interface for asking questions to a PDF document.
+
+## Setup
+
+```
+pip install -r requirements.txt
+```
+
+## Run
+
+```
+python main.py
+```
+
+A prompt will appear, where questions may be asked:
+
+```
+Query: How many locations does WeWork have?
+```
--- a/examples/langchain-document/main.py
+++ b/examples/langchain-document/main.py
@@ -0,0 +1,61 @@
+from langchain.document_loaders import OnlinePDFLoader
+from langchain.vectorstores import Chroma
+from langchain.embeddings import GPT4AllEmbeddings
+from langchain import PromptTemplate
+from langchain.llms import Ollama
+from langchain.callbacks.manager import CallbackManager
+from langchain.callbacks.streaming_stdout import StreamingStdOutCallbackHandler
+from langchain.chains import RetrievalQA
+import sys
+import os
+
+class SuppressStdout:
+    def __enter__(self):
+        self._original_stdout = sys.stdout
+        self._original_stderr = sys.stderr
+        sys.stdout = open(os.devnull, 'w')
+        sys.stderr = open(os.devnull, 'w')
+
+    def __exit__(self, exc_type, exc_val, exc_tb):
+        sys.stdout.close()
+        sys.stdout = self._original_stdout
+        sys.stderr = self._original_stderr
+
+# load the pdf and split it into chunks
+loader = OnlinePDFLoader("https://d18rn0p25nwr6d.cloudfront.net/CIK-0001813756/975b3e9b-268e-4798-a9e4-2a9a7c92dc10.pdf")
+data = loader.load()
+
+from langchain.text_splitter import RecursiveCharacterTextSplitter
+text_splitter = RecursiveCharacterTextSplitter(chunk_size=500, chunk_overlap=0)
+all_splits = text_splitter.split_documents(data)
+
+with SuppressStdout():
+    vectorstore = Chroma.from_documents(documents=all_splits, embedding=GPT4AllEmbeddings())
+
+while True:
+    query = input("\nQuery: ")
+    if query == "exit":
+        break
+    if query.strip() == "":
+        continue
+
+    # Prompt
+    template = """Use the following pieces of context to answer the question at the end. 
+    If you don't know the answer, just say that you don't know, don't try to make up an answer. 
+    Use three sentences maximum and keep the answer as concise as possible. 
+    {context}
+    Question: {question}
+    Helpful Answer:"""
+    QA_CHAIN_PROMPT = PromptTemplate(
+        input_variables=["context", "question"],
+        template=template,
+    )
+
+    llm = Ollama(model="llama2:13b", callback_manager=CallbackManager([StreamingStdOutCallbackHandler()]))
+    qa_chain = RetrievalQA.from_chain_type(
+        llm,
+        retriever=vectorstore.as_retriever(),
+        chain_type_kwargs={"prompt": QA_CHAIN_PROMPT},
+    )
+
+    result = qa_chain({"query": query})
--- a/examples/langchain-document/requirements.txt
+++ b/examples/langchain-document/requirements.txt
@@ -0,0 +1,109 @@
+absl-py==1.4.0
+aiohttp==3.8.5
+aiosignal==1.3.1
+anyio==3.7.1
+astunparse==1.6.3
+async-timeout==4.0.3
+attrs==23.1.0
+backoff==2.2.1
+beautifulsoup4==4.12.2
+bs4==0.0.1
+cachetools==5.3.1
+certifi==2023.7.22
+cffi==1.15.1
+chardet==5.2.0
+charset-normalizer==3.2.0
+Chroma==0.2.0
+chroma-hnswlib==0.7.2
+chromadb==0.4.5
+click==8.1.6
+coloredlogs==15.0.1
+cryptography==41.0.3
+dataclasses-json==0.5.14
+fastapi==0.99.1
+filetype==1.2.0
+flatbuffers==23.5.26
+frozenlist==1.4.0
+gast==0.4.0
+google-auth==2.22.0
+google-auth-oauthlib==1.0.0
+google-pasta==0.2.0
+gpt4all==1.0.8
+grpcio==1.57.0
+h11==0.14.0
+h5py==3.9.0
+httptools==0.6.0
+humanfriendly==10.0
+idna==3.4
+importlib-resources==6.0.1
+joblib==1.3.2
+keras==2.13.1
+langchain==0.0.261
+langsmith==0.0.21
+libclang==16.0.6
+lxml==4.9.3
+Markdown==3.4.4
+MarkupSafe==2.1.3
+marshmallow==3.20.1
+monotonic==1.6
+mpmath==1.3.0
+multidict==6.0.4
+mypy-extensions==1.0.0
+nltk==3.8.1
+numexpr==2.8.5
+numpy==1.24.3
+oauthlib==3.2.2
+onnxruntime==1.15.1
+openapi-schema-pydantic==1.2.4
+opt-einsum==3.3.0
+overrides==7.4.0
+packaging==23.1
+pdf2image==1.16.3
+pdfminer==20191125
+pdfminer.six==20221105
+Pillow==10.0.0
+posthog==3.0.1
+protobuf==4.24.0
+pulsar-client==3.2.0
+pyasn1==0.5.0
+pyasn1-modules==0.3.0
+pycparser==2.21
+pycryptodome==3.18.0
+pydantic==1.10.12
+PyPika==0.48.9
+python-dateutil==2.8.2
+python-dotenv==1.0.0
+python-magic==0.4.27
+PyYAML==6.0.1
+regex==2023.8.8
+requests==2.31.0
+requests-oauthlib==1.3.1
+rsa==4.9
+six==1.16.0
+sniffio==1.3.0
+soupsieve==2.4.1
+SQLAlchemy==2.0.19
+starlette==0.27.0
+sympy==1.12
+tabulate==0.9.0
+tenacity==8.2.2
+tensorboard==2.13.0
+tensorboard-data-server==0.7.1
+tensorflow==2.13.0
+tensorflow-estimator==2.13.0
+tensorflow-hub==0.14.0
+tensorflow-macos==2.13.0
+termcolor==2.3.0
+tokenizers==0.13.3
+tqdm==4.66.1
+typing-inspect==0.9.0
+typing_extensions==4.5.0
+unstructured==0.9.2
+urllib3==1.26.16
+uvicorn==0.23.2
+uvloop==0.17.0
+watchfiles==0.19.0
+websockets==11.0.3
+Werkzeug==2.3.6
+wrapt==1.15.0
+yarl==1.9.2
--- a/examples/langchain-web-summary/README.md
+++ b/examples/langchain-web-summary/README.md
@@ -0,0 +1,15 @@
+# LangChain Web Summarization
+
+This example summarizes a website
+
+## Setup
+
+```
+pip install -r requirements.txt
+```
+
+## Run
+
+```
+python main.py
+```
--- a/examples/langchain-web-summary/main.py
+++ b/examples/langchain-web-summary/main.py
@@ -0,0 +1,12 @@
+from langchain.llms import Ollama
+from langchain.document_loaders import WebBaseLoader
+from langchain.chains.summarize import load_summarize_chain
+
+loader = WebBaseLoader("https://ollama.ai/blog/run-llama2-uncensored-locally")
+docs = loader.load()
+
+llm = Ollama(model="llama2")
+chain = load_summarize_chain(llm, chain_type="stuff")
+
+result = chain.run(docs)
+print(result)
--- a/examples/langchain-web-summary/requirements.txt
+++ b/examples/langchain-web-summary/requirements.txt
@@ -0,0 +1,2 @@
+langchain==0.0.259
+bs4==0.0.1
--- a/examples/langchain/README.md
+++ b/examples/langchain/README.md
@@ -0,0 +1,21 @@
+# LangChain
+
+This example is a basic "hello world" of using LangChain with Ollama.
+
+## Setup
+
+```
+pip install -r requirements.txt
+```
+
+## Run
+
+```
+python main.py
+```
+
+Running this example will print the response for "hello":
+
+```
+Hello! It's nice to meet you. hopefully you are having a great day! Is there something I can help you with or would you like to chat?
+```
--- a/examples/langchain/main.py
+++ b/examples/langchain/main.py
@@ -0,0 +1,4 @@
+from langchain.llms import Ollama
+llm = Ollama(model="llama2")
+res = llm.predict("hello")
+print (res)
--- a/examples/langchain/requirements.txt
+++ b/examples/langchain/requirements.txt
@@ -0,0 +1 @@
+langchain==0.0.259
--- a/examples/mario/Modelfile
+++ b/examples/mario/Modelfile
@@ -0,0 +1,5 @@
+FROM llama2
+PARAMETER temperature 1
+SYSTEM """
+You are Mario from super mario bros, acting as an assistant.
+"""
--- a/examples/mario/logo.png
+++ b/examples/mario/logo.png
--- a/examples/mario/readme.md
+++ b/examples/mario/readme.md
@@ -0,0 +1,43 @@
+<img src="logo.png" alt="image of Italian plumber" height="200"/>
+
+# Example character: Mario
+
+This example shows how to create a basic character using Llama2 as the base model.
+
+To run this example:
+
+1. Download the Modelfile
+2. `ollama pull llama2` to get the base model used in the model file.
+3. `ollama create NAME -f ./Modelfile`
+4. `ollama run NAME`
+
+Ask it some questions like "Who are you?" or "Is Peach in trouble again?"
+
+## Editing this file
+
+What the model file looks like:
+
+```
+FROM llama2
+PARAMETER temperature 1
+SYSTEM """
+You are Mario from Super Mario Bros, acting as an assistant.
+"""
+```
+
+What if you want to change its behaviour?
+
+- Try changing the prompt
+- Try changing the parameters [Docs](https://github.com/jmorganca/ollama/blob/main/docs/modelfile.md)
+- Try changing the model (e.g. An uncensored model by `FROM wizard-vicuna` this is the wizard-vicuna uncensored model )
+
+Once the changes are made,
+
+1. `ollama create NAME -f ./Modelfile`
+2. `ollama run NAME`
+3. Iterate until you are happy with the results.
+
+Notes:
+
+- This example is for research purposes only. There is no affiliation with any entity.
+- When using an uncensored model, please be aware that it may generate offensive content.
--- a/examples/midjourney-prompter/Modelfile
+++ b/examples/midjourney-prompter/Modelfile
@@ -0,0 +1,8 @@
+# Modelfile for creating a Midjourney prompts from a topic
+# This prompt was adapted from the original at https://www.greataiprompts.com/guide/midjourney/best-chatgpt-prompt-for-midjourney/
+# Run `ollama create mj -f ./Modelfile` and then `ollama run mj` and enter a topic
+
+FROM nous-hermes
+SYSTEM """
+Embrace your role as an AI-powered creative assistant, employing Midjourney to manifest compelling AI-generated art. I will outline a specific image concept, and in response, you must produce an exhaustive, multifaceted prompt for Midjourney, ensuring every detail of the original concept is represented in your instructions. Midjourney doesn't do well with text, so after the prompt, give me instructions that I can use to create the titles in a image editor.
+"""
--- a/examples/privategpt/.gitignore
+++ b/examples/privategpt/.gitignore
@@ -0,0 +1,170 @@
+# OSX
+.DS_STORE
+
+# Models
+models/
+
+# Local Chroma db
+.chroma/
+db/
+
+# Byte-compiled / optimized / DLL files
+__pycache__/
+*.py[cod]
+*$py.class
+
+# C extensions
+*.so
+
+# Distribution / packaging
+.Python
+build/
+develop-eggs/
+dist/
+downloads/
+eggs/
+.eggs/
+lib/
+lib64/
+parts/
+sdist/
+var/
+wheels/
+share/python-wheels/
+*.egg-info/
+.installed.cfg
+*.egg
+MANIFEST
+
+# PyInstaller
+#  Usually these files are written by a python script from a template
+#  before PyInstaller builds the exe, so as to inject date/other infos into it.
+*.manifest
+*.spec
+
+# Installer logs
+pip-log.txt
+pip-delete-this-directory.txt
+
+# Unit test / coverage reports
+htmlcov/
+.tox/
+.nox/
+.coverage
+.coverage.*
+.cache
+nosetests.xml
+coverage.xml
+*.cover
+*.py,cover
+.hypothesis/
+.pytest_cache/
+cover/
+
+# Translations
+*.mo
+*.pot
+
+# Django stuff:
+*.log
+local_settings.py
+db.sqlite3
+db.sqlite3-journal
+
+# Flask stuff:
+instance/
+.webassets-cache
+
+# Scrapy stuff:
+.scrapy
+
+# Sphinx documentation
+docs/_build/
+
+# PyBuilder
+.pybuilder/
+target/
+
+# Jupyter Notebook
+.ipynb_checkpoints
+
+# IPython
+profile_default/
+ipython_config.py
+
+# pyenv
+#   For a library or package, you might want to ignore these files since the code is
+#   intended to run in multiple environments; otherwise, check them in:
+# .python-version
+
+# pipenv
+#   According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
+#   However, in case of collaboration, if having platform-specific dependencies or dependencies
+#   having no cross-platform support, pipenv may install dependencies that don't work, or not
+#   install all needed dependencies.
+#Pipfile.lock
+
+# poetry
+#   Similar to Pipfile.lock, it is generally recommended to include poetry.lock in version control.
+#   This is especially recommended for binary packages to ensure reproducibility, and is more
+#   commonly ignored for libraries.
+#   https://python-poetry.org/docs/basic-usage/#commit-your-poetrylock-file-to-version-control
+#poetry.lock
+
+# pdm
+#   Similar to Pipfile.lock, it is generally recommended to include pdm.lock in version control.
+#pdm.lock
+#   pdm stores project-wide configurations in .pdm.toml, but it is recommended to not include it
+#   in version control.
+#   https://pdm.fming.dev/#use-with-ide
+.pdm.toml
+
+# PEP 582; used by e.g. github.com/David-OConnor/pyflow and github.com/pdm-project/pdm
+__pypackages__/
+
+# Celery stuff
+celerybeat-schedule
+celerybeat.pid
+
+# SageMath parsed files
+*.sage.py
+
+# Environments
+.env
+.venv
+env/
+venv/
+ENV/
+env.bak/
+venv.bak/
+
+# Spyder project settings
+.spyderproject
+.spyproject
+
+# Rope project settings
+.ropeproject
+
+# mkdocs documentation
+/site
+
+# mypy
+.mypy_cache/
+.dmypy.json
+dmypy.json
+
+# Pyre type checker
+.pyre/
+
+# pytype static type analyzer
+.pytype/
+
+# Cython debug symbols
+cython_debug/
+
+# PyCharm
+#  JetBrains specific template is maintained in a separate JetBrains.gitignore that can
+#  be found at https://github.com/github/gitignore/blob/main/Global/JetBrains.gitignore
+#  and can be added to the global gitignore or merged into this file.  For a more nuclear
+#  option (not recommended) you can uncomment the following to ignore the entire idea folder.
+#.idea/
--- a/examples/privategpt/LICENSE
+++ b/examples/privategpt/LICENSE
@@ -0,0 +1,201 @@
+                                 Apache License
+                           Version 2.0, January 2004
+                        http://www.apache.org/licenses/
+
+   TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
+
+   1. Definitions.
+
+      "License" shall mean the terms and conditions for use, reproduction,
+      and distribution as defined by Sections 1 through 9 of this document.
+
+      "Licensor" shall mean the copyright owner or entity authorized by
+      the copyright owner that is granting the License.
+
+      "Legal Entity" shall mean the union of the acting entity and all
+      other entities that control, are controlled by, or are under common
+      control with that entity. For the purposes of this definition,
+      "control" means (i) the power, direct or indirect, to cause the
+      direction or management of such entity, whether by contract or
+      otherwise, or (ii) ownership of fifty percent (50%) or more of the
+      outstanding shares, or (iii) beneficial ownership of such entity.
+
+      "You" (or "Your") shall mean an individual or Legal Entity
+      exercising permissions granted by this License.
+
+      "Source" form shall mean the preferred form for making modifications,
+      including but not limited to software source code, documentation
+      source, and configuration files.
+
+      "Object" form shall mean any form resulting from mechanical
+      transformation or translation of a Source form, including but
+      not limited to compiled object code, generated documentation,
+      and conversions to other media types.
+
+      "Work" shall mean the work of authorship, whether in Source or
+      Object form, made available under the License, as indicated by a
+      copyright notice that is included in or attached to the work
+      (an example is provided in the Appendix below).
+
+      "Derivative Works" shall mean any work, whether in Source or Object
+      form, that is based on (or derived from) the Work and for which the
+      editorial revisions, annotations, elaborations, or other modifications
+      represent, as a whole, an original work of authorship. For the purposes
+      of this License, Derivative Works shall not include works that remain
+      separable from, or merely link (or bind by name) to the interfaces of,
+      the Work and Derivative Works thereof.
+
+      "Contribution" shall mean any work of authorship, including
+      the original version of the Work and any modifications or additions
+      to that Work or Derivative Works thereof, that is intentionally
+      submitted to Licensor for inclusion in the Work by the copyright owner
+      or by an individual or Legal Entity authorized to submit on behalf of
+      the copyright owner. For the purposes of this definition, "submitted"
+      means any form of electronic, verbal, or written communication sent
+      to the Licensor or its representatives, including but not limited to
+      communication on electronic mailing lists, source code control systems,
+      and issue tracking systems that are managed by, or on behalf of, the
+      Licensor for the purpose of discussing and improving the Work, but
+      excluding communication that is conspicuously marked or otherwise
+      designated in writing by the copyright owner as "Not a Contribution."
+
+      "Contributor" shall mean Licensor and any individual or Legal Entity
+      on behalf of whom a Contribution has been received by Licensor and
+      subsequently incorporated within the Work.
+
+   2. Grant of Copyright License. Subject to the terms and conditions of
+      this License, each Contributor hereby grants to You a perpetual,
+      worldwide, non-exclusive, no-charge, royalty-free, irrevocable
+      copyright license to reproduce, prepare Derivative Works of,
+      publicly display, publicly perform, sublicense, and distribute the
+      Work and such Derivative Works in Source or Object form.
+
+   3. Grant of Patent License. Subject to the terms and conditions of
+      this License, each Contributor hereby grants to You a perpetual,
+      worldwide, non-exclusive, no-charge, royalty-free, irrevocable
+      (except as stated in this section) patent license to make, have made,
+      use, offer to sell, sell, import, and otherwise transfer the Work,
+      where such license applies only to those patent claims licensable
+      by such Contributor that are necessarily infringed by their
+      Contribution(s) alone or by combination of their Contribution(s)
+      with the Work to which such Contribution(s) was submitted. If You
+      institute patent litigation against any entity (including a
+      cross-claim or counterclaim in a lawsuit) alleging that the Work
+      or a Contribution incorporated within the Work constitutes direct
+      or contributory patent infringement, then any patent licenses
+      granted to You under this License for that Work shall terminate
+      as of the date such litigation is filed.
+
+   4. Redistribution. You may reproduce and distribute copies of the
+      Work or Derivative Works thereof in any medium, with or without
+      modifications, and in Source or Object form, provided that You
+      meet the following conditions:
+
+      (a) You must give any other recipients of the Work or
+          Derivative Works a copy of this License; and
+
+      (b) You must cause any modified files to carry prominent notices
+          stating that You changed the files; and
+
+      (c) You must retain, in the Source form of any Derivative Works
+          that You distribute, all copyright, patent, trademark, and
+          attribution notices from the Source form of the Work,
+          excluding those notices that do not pertain to any part of
+          the Derivative Works; and
+
+      (d) If the Work includes a "NOTICE" text file as part of its
+          distribution, then any Derivative Works that You distribute must
+          include a readable copy of the attribution notices contained
+          within such NOTICE file, excluding those notices that do not
+          pertain to any part of the Derivative Works, in at least one
+          of the following places: within a NOTICE text file distributed
+          as part of the Derivative Works; within the Source form or
+          documentation, if provided along with the Derivative Works; or,
+          within a display generated by the Derivative Works, if and
+          wherever such third-party notices normally appear. The contents
+          of the NOTICE file are for informational purposes only and
+          do not modify the License. You may add Your own attribution
+          notices within Derivative Works that You distribute, alongside
+          or as an addendum to the NOTICE text from the Work, provided
+          that such additional attribution notices cannot be construed
+          as modifying the License.
+
+      You may add Your own copyright statement to Your modifications and
+      may provide additional or different license terms and conditions
+      for use, reproduction, or distribution of Your modifications, or
+      for any such Derivative Works as a whole, provided Your use,
+      reproduction, and distribution of the Work otherwise complies with
+      the conditions stated in this License.
+
+   5. Submission of Contributions. Unless You explicitly state otherwise,
+      any Contribution intentionally submitted for inclusion in the Work
+      by You to the Licensor shall be under the terms and conditions of
+      this License, without any additional terms or conditions.
+      Notwithstanding the above, nothing herein shall supersede or modify
+      the terms of any separate license agreement you may have executed
+      with Licensor regarding such Contributions.
+
+   6. Trademarks. This License does not grant permission to use the trade
+      names, trademarks, service marks, or product names of the Licensor,
+      except as required for reasonable and customary use in describing the
+      origin of the Work and reproducing the content of the NOTICE file.
+
+   7. Disclaimer of Warranty. Unless required by applicable law or
+      agreed to in writing, Licensor provides the Work (and each
+      Contributor provides its Contributions) on an "AS IS" BASIS,
+      WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
+      implied, including, without limitation, any warranties or conditions
+      of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
+      PARTICULAR PURPOSE. You are solely responsible for determining the
+      appropriateness of using or redistributing the Work and assume any
+      risks associated with Your exercise of permissions under this License.
+
+   8. Limitation of Liability. In no event and under no legal theory,
+      whether in tort (including negligence), contract, or otherwise,
+      unless required by applicable law (such as deliberate and grossly
+      negligent acts) or agreed to in writing, shall any Contributor be
+      liable to You for damages, including any direct, indirect, special,
+      incidental, or consequential damages of any character arising as a
+      result of this License or out of the use or inability to use the
+      Work (including but not limited to damages for loss of goodwill,
+      work stoppage, computer failure or malfunction, or any and all
+      other commercial damages or losses), even if such Contributor
+      has been advised of the possibility of such damages.
+
+   9. Accepting Warranty or Additional Liability. While redistributing
+      the Work or Derivative Works thereof, You may choose to offer,
+      and charge a fee for, acceptance of support, warranty, indemnity,
+      or other liability obligations and/or rights consistent with this
+      License. However, in accepting such obligations, You may act only
+      on Your own behalf and on Your sole responsibility, not on behalf
+      of any other Contributor, and only if You agree to indemnify,
+      defend, and hold each Contributor harmless for any liability
+      incurred by, or claims asserted against, such Contributor by reason
+      of your accepting any such warranty or additional liability.
+
+   END OF TERMS AND CONDITIONS
+
+   APPENDIX: How to apply the Apache License to your work.
+
+      To apply the Apache License to your work, attach the following
+      boilerplate notice, with the fields enclosed by brackets "[]"
+      replaced with your own identifying information. (Don't include
+      the brackets!)  The text should be enclosed in the appropriate
+      comment syntax for the file format. We also recommend that a
+      file or class name and description of purpose be included on the
+      same "printed page" as the copyright notice for easier
+      identification within third-party archives.
+
+   Copyright [yyyy] [name of copyright owner]
+
+   Licensed under the Apache License, Version 2.0 (the "License");
+   you may not use this file except in compliance with the License.
+   You may obtain a copy of the License at
+
+       http://www.apache.org/licenses/LICENSE-2.0
+
+   Unless required by applicable law or agreed to in writing, software
+   distributed under the License is distributed on an "AS IS" BASIS,
+   WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+   See the License for the specific language governing permissions and
+   limitations under the License.
--- a/examples/privategpt/README.md
+++ b/examples/privategpt/README.md
@@ -0,0 +1,91 @@
+# PrivateGPT with Llama 2 uncensored
+
+https://github.com/jmorganca/ollama/assets/3325447/20cf8ec6-ff25-42c6-bdd8-9be594e3ce1b
+
+> Note: this example is a slightly modified version of PrivateGPT using models such as Llama 2 Uncensored. All credit for PrivateGPT goes to Iván Martínez who is the creator of it, and you can find his GitHub repo [here](https://github.com/imartinez/privateGPT).
+
+### Setup
+
+Set up a virtual environment (optional):
+
+```
+python3 -m venv .venv
+source .venv/bin/activate
+```
+
+Install the Python dependencies:
+
+```shell
+pip install -r requirements.txt
+```
+
+Pull the model you'd like to use:
+
+```
+ollama pull llama2-uncensored
+```
+
+### Getting WeWork's latest quarterly earnings report (10-Q)
+
+```
+mkdir source_documents
+curl https://d18rn0p25nwr6d.cloudfront.net/CIK-0001813756/975b3e9b-268e-4798-a9e4-2a9a7c92dc10.pdf -o source_documents/wework.pdf
+```
+
+### Ingesting files
+
+```shell
+python ingest.py
+```
+
+Output should look like this:
+
+```shell
+Creating new vectorstore
+Loading documents from source_documents
+Loading new documents: 100%|██████████████████████| 1/1 [00:01<00:00,  1.73s/it]
+Loaded 1 new documents from source_documents
+Split into 90 chunks of text (max. 500 tokens each)
+Creating embeddings. May take some minutes...
+Using embedded DuckDB with persistence: data will be stored in: db
+Ingestion complete! You can now run privateGPT.py to query your documents
+```
+
+### Ask questions
+
+```shell
+python privateGPT.py
+
+Enter a query: How many locations does WeWork have?
+
+> Answer (took 17.7 s.):
+As of June 2023, WeWork has 777 locations worldwide, including 610 Consolidated Locations (as defined in the section entitled Key Performance Indicators).
+```
+
+### Try a different model:
+
+```
+ollama pull llama2:13b
+MODEL=llama2:13b python privateGPT.py
+```
+
+## Adding more files
+
+Put any and all your files into the `source_documents` directory
+
+The supported extensions are:
+
+- `.csv`: CSV,
+- `.docx`: Word Document,
+- `.doc`: Word Document,
+- `.enex`: EverNote,
+- `.eml`: Email,
+- `.epub`: EPub,
+- `.html`: HTML File,
+- `.md`: Markdown,
+- `.msg`: Outlook Message,
+- `.odt`: Open Document Text,
+- `.pdf`: Portable Document Format (PDF),
+- `.pptx` : PowerPoint Document,
+- `.ppt` : PowerPoint Document,
+- `.txt`: Text file (UTF-8),
--- a/examples/privategpt/constants.py
+++ b/examples/privategpt/constants.py
@@ -0,0 +1,12 @@
+import os
+from chromadb.config import Settings
+
+# Define the folder for storing database
+PERSIST_DIRECTORY = os.environ.get('PERSIST_DIRECTORY', 'db')
+
+# Define the Chroma settings
+CHROMA_SETTINGS = Settings(
+        chroma_db_impl='duckdb+parquet',
+        persist_directory=PERSIST_DIRECTORY,
+        anonymized_telemetry=False
+)
--- a/examples/privategpt/ingest.py
+++ b/examples/privategpt/ingest.py
@@ -0,0 +1,161 @@
+#!/usr/bin/env python3
+import os
+import glob
+from typing import List
+from multiprocessing import Pool
+from tqdm import tqdm
+
+from langchain.document_loaders import (
+    CSVLoader,
+    EverNoteLoader,
+    PyMuPDFLoader,
+    TextLoader,
+    UnstructuredEmailLoader,
+    UnstructuredEPubLoader,
+    UnstructuredHTMLLoader,
+    UnstructuredMarkdownLoader,
+    UnstructuredODTLoader,
+    UnstructuredPowerPointLoader,
+    UnstructuredWordDocumentLoader,
+)
+
+from langchain.text_splitter import RecursiveCharacterTextSplitter
+from langchain.vectorstores import Chroma
+from langchain.embeddings import HuggingFaceEmbeddings
+from langchain.docstore.document import Document
+from constants import CHROMA_SETTINGS
+
+
+# Load environment variables
+persist_directory = os.environ.get('PERSIST_DIRECTORY', 'db')
+source_directory = os.environ.get('SOURCE_DIRECTORY', 'source_documents')
+embeddings_model_name = os.environ.get('EMBEDDINGS_MODEL_NAME', 'all-MiniLM-L6-v2')
+chunk_size = 500
+chunk_overlap = 50
+
+# Custom document loaders
+class MyElmLoader(UnstructuredEmailLoader):
+    """Wrapper to fallback to text/plain when default does not work"""
+
+    def load(self) -> List[Document]:
+        """Wrapper adding fallback for elm without html"""
+        try:
+            try:
+                doc = UnstructuredEmailLoader.load(self)
+            except ValueError as e:
+                if 'text/html content not found in email' in str(e):
+                    # Try plain text
+                    self.unstructured_kwargs["content_source"]="text/plain"
+                    doc = UnstructuredEmailLoader.load(self)
+                else:
+                    raise
+        except Exception as e:
+            # Add file_path to exception message
+            raise type(e)(f"{self.file_path}: {e}") from e
+
+        return doc
+
+
+# Map file extensions to document loaders and their arguments
+LOADER_MAPPING = {
+    ".csv": (CSVLoader, {}),
+    # ".docx": (Docx2txtLoader, {}),
+    ".doc": (UnstructuredWordDocumentLoader, {}),
+    ".docx": (UnstructuredWordDocumentLoader, {}),
+    ".enex": (EverNoteLoader, {}),
+    ".eml": (MyElmLoader, {}),
+    ".epub": (UnstructuredEPubLoader, {}),
+    ".html": (UnstructuredHTMLLoader, {}),
+    ".md": (UnstructuredMarkdownLoader, {}),
+    ".odt": (UnstructuredODTLoader, {}),
+    ".pdf": (PyMuPDFLoader, {}),
+    ".ppt": (UnstructuredPowerPointLoader, {}),
+    ".pptx": (UnstructuredPowerPointLoader, {}),
+    ".txt": (TextLoader, {"encoding": "utf8"}),
+    # Add more mappings for other file extensions and loaders as needed
+}
+
+
+def load_single_document(file_path: str) -> List[Document]:
+    ext = "." + file_path.rsplit(".", 1)[-1]
+    if ext in LOADER_MAPPING:
+        loader_class, loader_args = LOADER_MAPPING[ext]
+        loader = loader_class(file_path, **loader_args)
+        return loader.load()
+
+    raise ValueError(f"Unsupported file extension '{ext}'")
+
+def load_documents(source_dir: str, ignored_files: List[str] = []) -> List[Document]:
+    """
+    Loads all documents from the source documents directory, ignoring specified files
+    """
+    all_files = []
+    for ext in LOADER_MAPPING:
+        all_files.extend(
+            glob.glob(os.path.join(source_dir, f"**/*{ext}"), recursive=True)
+        )
+    filtered_files = [file_path for file_path in all_files if file_path not in ignored_files]
+
+    with Pool(processes=os.cpu_count()) as pool:
+        results = []
+        with tqdm(total=len(filtered_files), desc='Loading new documents', ncols=80) as pbar:
+            for i, docs in enumerate(pool.imap_unordered(load_single_document, filtered_files)):
+                results.extend(docs)
+                pbar.update()
+
+    return results
+
+def process_documents(ignored_files: List[str] = []) -> List[Document]:
+    """
+    Load documents and split in chunks
+    """
+    print(f"Loading documents from {source_directory}")
+    documents = load_documents(source_directory, ignored_files)
+    if not documents:
+        print("No new documents to load")
+        exit(0)
+    print(f"Loaded {len(documents)} new documents from {source_directory}")
+    text_splitter = RecursiveCharacterTextSplitter(chunk_size=chunk_size, chunk_overlap=chunk_overlap)
+    texts = text_splitter.split_documents(documents)
+    print(f"Split into {len(texts)} chunks of text (max. {chunk_size} tokens each)")
+    return texts
+
+def does_vectorstore_exist(persist_directory: str) -> bool:
+    """
+    Checks if vectorstore exists
+    """
+    if os.path.exists(os.path.join(persist_directory, 'index')):
+        if os.path.exists(os.path.join(persist_directory, 'chroma-collections.parquet')) and os.path.exists(os.path.join(persist_directory, 'chroma-embeddings.parquet')):
+            list_index_files = glob.glob(os.path.join(persist_directory, 'index/*.bin'))
+            list_index_files += glob.glob(os.path.join(persist_directory, 'index/*.pkl'))
+            # At least 3 documents are needed in a working vectorstore
+            if len(list_index_files) > 3:
+                return True
+    return False
+
+def main():
+    # Create embeddings
+    embeddings = HuggingFaceEmbeddings(model_name=embeddings_model_name)
+
+    if does_vectorstore_exist(persist_directory):
+        # Update and store locally vectorstore
+        print(f"Appending to existing vectorstore at {persist_directory}")
+        db = Chroma(persist_directory=persist_directory, embedding_function=embeddings, client_settings=CHROMA_SETTINGS)
+        collection = db.get()
+        texts = process_documents([metadata['source'] for metadata in collection['metadatas']])
+        print(f"Creating embeddings. May take some minutes...")
+        db.add_documents(texts)
+    else:
+        # Create and store locally vectorstore
+        print("Creating new vectorstore")
+        texts = process_documents()
+        print(f"Creating embeddings. May take some minutes...")
+        db = Chroma.from_documents(texts, embeddings, persist_directory=persist_directory, client_settings=CHROMA_SETTINGS)
+    db.persist()
+    db = None
+
+    print(f"Ingestion complete! You can now run privateGPT.py to query your documents")
+
+
+if __name__ == "__main__":
+    main()
--- a/examples/privategpt/poetry.lock
+++ b/examples/privategpt/poetry.lock
--- a/examples/privategpt/privateGPT.py
+++ b/examples/privategpt/privateGPT.py
@@ -0,0 +1,71 @@
+#!/usr/bin/env python3
+from langchain.chains import RetrievalQA
+from langchain.embeddings import HuggingFaceEmbeddings
+from langchain.callbacks.streaming_stdout import StreamingStdOutCallbackHandler
+from langchain.vectorstores import Chroma
+from langchain.llms import Ollama
+import os
+import argparse
+import time
+
+model = os.environ.get("MODEL", "llama2-uncensored")
+# For embeddings model, the example uses a sentence-transformers model
+# https://www.sbert.net/docs/pretrained_models.html 
+# "The all-mpnet-base-v2 model provides the best quality, while all-MiniLM-L6-v2 is 5 times faster and still offers good quality."
+embeddings_model_name = os.environ.get("EMBEDDINGS_MODEL_NAME", "all-MiniLM-L6-v2")
+persist_directory = os.environ.get("PERSIST_DIRECTORY", "db")
+target_source_chunks = int(os.environ.get('TARGET_SOURCE_CHUNKS',4))
+
+from constants import CHROMA_SETTINGS
+
+def main():
+    # Parse the command line arguments
+    args = parse_arguments()
+    embeddings = HuggingFaceEmbeddings(model_name=embeddings_model_name)
+    db = Chroma(persist_directory=persist_directory, embedding_function=embeddings, client_settings=CHROMA_SETTINGS)
+    retriever = db.as_retriever(search_kwargs={"k": target_source_chunks})
+    # activate/deactivate the streaming StdOut callback for LLMs
+    callbacks = [] if args.mute_stream else [StreamingStdOutCallbackHandler()]
+
+    llm = Ollama(model=model, callbacks=callbacks)
+
+    qa = RetrievalQA.from_chain_type(llm=llm, chain_type="stuff", retriever=retriever, return_source_documents= not args.hide_source)
+    # Interactive questions and answers
+    while True:
+        query = input("\nEnter a query: ")
+        if query == "exit":
+            break
+        if query.strip() == "":
+            continue
+
+        # Get the answer from the chain
+        start = time.time()
+        res = qa(query)
+        answer, docs = res['result'], [] if args.hide_source else res['source_documents']
+        end = time.time()
+
+        # Print the result
+        print("\n\n> Question:")
+        print(query)
+        print(answer)
+
+        # Print the relevant sources used for the answer
+        for document in docs:
+            print("\n> " + document.metadata["source"] + ":")
+            print(document.page_content)
+
+def parse_arguments():
+    parser = argparse.ArgumentParser(description='privateGPT: Ask questions to your documents without an internet connection, '
+                                                 'using the power of LLMs.')
+    parser.add_argument("--hide-source", "-S", action='store_true',
+                        help='Use this flag to disable printing of source documents used for answers.')
+
+    parser.add_argument("--mute-stream", "-M",
+                        action='store_true',
+                        help='Use this flag to disable the streaming StdOut callback for LLMs.')
+
+    return parser.parse_args()
+
+
+if __name__ == "__main__":
+    main()
--- a/examples/privategpt/pyproject.toml
+++ b/examples/privategpt/pyproject.toml
@@ -0,0 +1,26 @@
+[tool.poetry]
+name = "privategpt"
+version = "0.1.0"
+description = ""
+authors = ["Ivan Martinez <ivanmartit@gmail.com>"]
+license = "Apache Version 2.0"
+readme = "README.md"
+
+[tool.poetry.dependencies]
+python = "^3.10"
+langchain = "0.0.261"
+gpt4all = "^1.0.3"
+chromadb = "^0.3.26"
+PyMuPDF = "^1.22.5"
+python-dotenv = "^1.0.0"
+unstructured = "^0.8.0"
+extract-msg = "^0.41.5"
+tabulate = "^0.9.0"
+pandoc = "^2.3"
+pypandoc = "^1.11"
+tqdm = "^4.65.0"
+sentence-transformers = "^2.2.2"
+
+[build-system]
+requires = ["poetry-core"]
+build-backend = "poetry.core.masonry.api"
--- a/examples/privategpt/requirements.txt
+++ b/examples/privategpt/requirements.txt
--- a/examples/python/README.md
+++ b/examples/python/README.md
@@ -1,15 +0,0 @@
-# Python
-
-This is a simple example of calling the Ollama api from a python app.
-
-First, download a model:
-
-```
-curl -L https://huggingface.co/TheBloke/orca_mini_3B-GGML/resolve/main/orca-mini-3b.ggmlv3.q4_1.bin -o orca.bin
-```
-
-Then run it using the example script. You'll need to have Ollama running on your machine.
-
-```
-python3 main.py orca.bin
-```
--- a/examples/python/client.py
+++ b/examples/python/client.py
@@ -0,0 +1,38 @@
+import json
+import requests
+
+# NOTE: ollama must be running for this to work, start the ollama app or run `ollama serve`
+model = 'llama2' # TODO: update this for whatever model you wish to use
+
+def generate(prompt, context):
+    r = requests.post('http://localhost:11434/api/generate',
+                      json={
+                          'model': model,
+                          'prompt': prompt,
+                          'context': context,
+                      },
+                      stream=True)
+    r.raise_for_status()
+
+    for line in r.iter_lines():
+        body = json.loads(line)
+        response_part = body.get('response', '')
+        # the response streams one token at a time, print that as we recieve it
+        print(response_part, end='', flush=True)
+
+        if 'error' in body:
+            raise Exception(body['error'])
+
+        if body.get('done', False):
+            return body['context']
+
+def main():
+    context = [] # the context stores a conversation history, you can use this to make the model more context aware
+    while True:
+        user_input = input("Enter a prompt: ")
+        print()
+        context = generate(user_input, context)
+        print()
+
+if __name__ == "__main__":
+    main()
--- a/examples/python/main.py
+++ b/examples/python/main.py
@@ -1,32 +0,0 @@
-import http.client
-import json
-import os
-import sys
-
-if len(sys.argv) < 2:
-    print("Usage: python main.py <model file>")
-    sys.exit(1)
-
-conn = http.client.HTTPConnection('localhost', 11434)
-
-headers = { 'Content-Type': 'application/json' }
-
-# generate text from the model
-conn.request("POST", "/api/generate", json.dumps({
-    'model': os.path.join(os.getcwd(), sys.argv[1]),
-    'prompt': 'write me a short story',
-    'stream': True
-}), headers)
-
-response = conn.getresponse()
-
-def parse_generate(data):
-    for event in data.decode('utf-8').split("\n"):
-        if not event:
-            continue
-        yield event
-
-if response.status == 200:
-    for chunk in response:
-        for event in parse_generate(chunk):
-            print(json.loads(event)['response'], end="", flush=True)
--- a/examples/recipemaker/Modelfile
+++ b/examples/recipemaker/Modelfile
@@ -0,0 +1,6 @@
+# Modelfile for creating a recipe from a list of ingredients
+# Run `ollama create recipemaker -f ./Modelfile` and then `ollama run recipemaker` and feed it lists of ingredients to create recipes around.
+FROM nous-hermes
+SYSTEM """
+The instruction will be a list of ingredients. You should generate a recipe that can be made in less than an hour. You can also include ingredients that most people will find in their pantry every day. The recipe should be 4 people and you should include a description of what the meal will taste like
+"""
--- a/examples/sentiments/Modelfile
+++ b/examples/sentiments/Modelfile
@@ -0,0 +1,28 @@
+# Modelfile for creating a sentiment analyzer. 
+# Run `ollama create sentiments -f pathtofile` and then `ollama run sentiments` and enter a topic
+
+FROM orca
+TEMPLATE """
+{{- if .First }}
+### System:
+{{ .System }}
+{{- end }}
+### User: 
+I hate it when my phone dies
+### Response: 
+NEGATIVE
+### User: 
+He is awesome
+### Response: 
+POSITIVE
+### User: 
+This is the link to the article
+### Response: 
+NEUTRAL
+### User:
+{{ .Prompt }}
+
+### Response:
+"""
+
+SYSTEM """You are a sentiment analyzer. You will receive text and output only one word, either POSITIVE or NEGATIVE or NEUTRAL, depending on the sentiment of the text."""
--- a/examples/sentiments/Readme.md
+++ b/examples/sentiments/Readme.md
@@ -0,0 +1,25 @@
+# Sentiments Modelfile
+
+This is a simple sentiments analyzer using the Orca model. When you pull Orca from the registry, it has a Template already defined that looks like this:
+
+```Modelfile
+{{- if .First }}
+### System:
+{{ .System }}
+{{- end }}
+
+### User:
+{{ .Prompt }}
+
+### Response:
+```
+
+If we just wanted to have the text:
+
+```Plaintext
+You are a sentiment analyzer. You will receive text and output only one word, either POSITIVE or NEGATIVE or NEUTRAL, depending on the sentiment of the text.
+```
+
+then we could have put this in a SYSTEM block. But we want to provide examples which require updating the full Template. Any Modelfile you create will inherit all the settings from the source model. But in this example, we are overriding the Template.
+
+When providing examples for the input and output, you should include the way the model usually provides information. Since the Orca model expects a user prompt to appear after ### User: and the response is after ### Response, we should format our examples like that as well. If we were using the Llama 2 model, the format would be a bit different.
--- a/examples/tweetwriter/Modelfile
+++ b/examples/tweetwriter/Modelfile
@@ -0,0 +1,7 @@
+# Modelfile for creating a tweet from a topic
+# Run `ollama create tweetwriter -f ./Modelfile` and then `ollama run tweetwriter` and enter a topic
+
+FROM nous-hermes
+SYSTEM """
+You are a content marketer who needs to come up with a short but succinct tweet. Make sure to include the appropriate hashtags and links. Sometimes when appropriate, describe a meme that can be included as well. All answers should be in the form of a tweet which has a max size of 280 characters. Every instruction will be the topic to create a tweet about.
+"""
--- a/format/openssh.go
+++ b/format/openssh.go
@@ -0,0 +1,102 @@
+// Copyright 2012 The Go Authors. All rights reserved.
+// Use of this source code is governed by a BSD-style
+// license that can be found in the LICENSE file.
+
+// Code originally from https://go-review.googlesource.com/c/crypto/+/218620
+
+// TODO: replace with upstream once the above change is merged and released.
+
+package format
+
+import (
+	"crypto"
+	"crypto/ed25519"
+	"crypto/rand"
+	"encoding/binary"
+	"encoding/pem"
+	"fmt"
+
+	"golang.org/x/crypto/ssh"
+)
+
+const privateKeyAuthMagic = "openssh-key-v1\x00"
+
+type openSSHEncryptedPrivateKey struct {
+	CipherName string
+	KDFName    string
+	KDFOptions string
+	KeysCount  uint32
+	PubKey     []byte
+	KeyBlocks  []byte
+}
+
+type openSSHPrivateKey struct {
+	Check1  uint32
+	Check2  uint32
+	Keytype string
+	Rest    []byte `ssh:"rest"`
+}
+
+type openSSHEd25519PrivateKey struct {
+	Pub     []byte
+	Priv    []byte
+	Comment string
+	Pad     []byte `ssh:"rest"`
+}
+
+func OpenSSHPrivateKey(key crypto.PrivateKey, comment string) (*pem.Block, error) {
+	var check uint32
+	if err := binary.Read(rand.Reader, binary.BigEndian, &check); err != nil {
+		return nil, err
+	}
+
+	var pk1 openSSHPrivateKey
+	pk1.Check1 = check
+	pk1.Check2 = check
+
+	var w openSSHEncryptedPrivateKey
+	w.KeysCount = 1
+
+	if k, ok := key.(*ed25519.PrivateKey); ok {
+		key = *k
+	}
+
+	switch k := key.(type) {
+	case ed25519.PrivateKey:
+		pub, priv := k[32:], k
+		key := openSSHEd25519PrivateKey{
+			Pub:     pub,
+			Priv:    priv,
+			Comment: comment,
+		}
+
+		pk1.Keytype = ssh.KeyAlgoED25519
+		pk1.Rest = ssh.Marshal(key)
+
+		w.PubKey = ssh.Marshal(struct {
+			KeyType string
+			Pub     []byte
+		}{
+			ssh.KeyAlgoED25519, pub,
+		})
+	default:
+		return nil, fmt.Errorf("ssh: unknown key type %T", k)
+	}
+
+	w.KeyBlocks = openSSHPadding(ssh.Marshal(pk1), 8)
+
+	w.CipherName, w.KDFName, w.KDFOptions = "none", "none", ""
+
+	return &pem.Block{
+		Type:  "OPENSSH PRIVATE KEY",
+		Bytes: append([]byte(privateKeyAuthMagic), ssh.Marshal(w)...),
+	}, nil
+}
+
+func openSSHPadding(block []byte, blocksize int) []byte {
+	for i, j := 0, len(block); (j+i)%blocksize != 0; i++ {
+		block = append(block, byte(i+1))
+	}
+
+	return block
+}
--- a/format/time.go
+++ b/format/time.go
@@ -0,0 +1,141 @@
+package format
+
+import (
+	"fmt"
+	"math"
+	"strings"
+	"time"
+)
+
+// HumanDuration returns a human-readable approximation of a duration
+// (eg. "About a minute", "4 hours ago", etc.).
+// Modified version of github.com/docker/go-units.HumanDuration
+func HumanDuration(d time.Duration) string {
+	return HumanDurationWithCase(d, true)
+}
+
+// HumanDurationWithCase returns a human-readable approximation of a
+// duration (eg. "About a minute", "4 hours ago", etc.). but allows
+// you to specify whether the first word should be capitalized
+// (eg. "About" vs. "about")
+func HumanDurationWithCase(d time.Duration, useCaps bool) string {
+	seconds := int(d.Seconds())
+
+	switch {
+	case seconds < 1:
+		if useCaps {
+			return "Less than a second"
+		}
+		return "less than a second"
+	case seconds == 1:
+		return "1 second"
+	case seconds < 60:
+		return fmt.Sprintf("%d seconds", seconds)
+	}
+
+	minutes := int(d.Minutes())
+	switch {
+	case minutes == 1:
+		if useCaps {
+			return "About a minute"
+		}
+		return "about a minute"
+	case minutes < 60:
+		return fmt.Sprintf("%d minutes", minutes)
+	}
+
+	hours := int(math.Round(d.Hours()))
+	switch {
+	case hours == 1:
+		if useCaps {
+			return "About an hour"
+		}
+		return "about an hour"
+	case hours < 48:
+		return fmt.Sprintf("%d hours", hours)
+	case hours < 24*7*2:
+		return fmt.Sprintf("%d days", hours/24)
+	case hours < 24*30*2:
+		return fmt.Sprintf("%d weeks", hours/24/7)
+	case hours < 24*365*2:
+		return fmt.Sprintf("%d months", hours/24/30)
+	}
+
+	return fmt.Sprintf("%d years", int(d.Hours())/24/365)
+}
+
+func HumanTime(t time.Time, zeroValue string) string {
+	return humanTimeWithCase(t, zeroValue, true)
+}
+
+func HumanTimeLower(t time.Time, zeroValue string) string {
+	return humanTimeWithCase(t, zeroValue, false)
+}
+
+func humanTimeWithCase(t time.Time, zeroValue string, useCaps bool) string {
+	if t.IsZero() {
+		return zeroValue
+	}
+
+	delta := time.Since(t)
+	if delta < 0 {
+		return HumanDurationWithCase(-delta, useCaps) + " from now"
+	}
+	return HumanDurationWithCase(delta, useCaps) + " ago"
+}
+
+// ExcatDuration returns a human readable hours/minutes/seconds or milliseconds format of a duration
+// the most precise level of duration is milliseconds
+func ExactDuration(d time.Duration) string {
+	if d.Seconds() < 1 {
+		if d.Milliseconds() == 1 {
+			return fmt.Sprintf("%d millisecond", d.Milliseconds())
+		}
+		return fmt.Sprintf("%d milliseconds", d.Milliseconds())
+	}
+
+	var readableDur strings.Builder
+
+	dur := d.String()
+
+	// split the default duration string format of 0h0m0s into something nicer to read
+	h := strings.Split(dur, "h")
+	if len(h) > 1 {
+		hours := h[0]
+		if hours == "1" {
+			readableDur.WriteString(fmt.Sprintf("%s hour ", hours))
+		} else {
+			readableDur.WriteString(fmt.Sprintf("%s hours ", hours))
+		}
+		dur = h[1]
+	}
+
+	m := strings.Split(dur, "m")
+	if len(m) > 1 {
+		mins := m[0]
+		switch mins {
+		case "0":
+			// skip
+		case "1":
+			readableDur.WriteString(fmt.Sprintf("%s minute ", mins))
+		default:
+			readableDur.WriteString(fmt.Sprintf("%s minutes ", mins))
+		}
+		dur = m[1]
+	}
+
+	s := strings.Split(dur, "s")
+	if len(s) > 0 {
+		sec := s[0]
+		switch sec {
+		case "0":
+			// skip
+		case "1":
+			readableDur.WriteString(fmt.Sprintf("%s second ", sec))
+		default:
+			readableDur.WriteString(fmt.Sprintf("%s seconds ", sec))
+		}
+	}
+
+	return strings.TrimSpace(readableDur.String())
+}
--- a/format/time_test.go
+++ b/format/time_test.go
@@ -0,0 +1,102 @@
+package format
+
+import (
+	"testing"
+	"time"
+)
+
+func assertEqual(t *testing.T, a interface{}, b interface{}) {
+	if a != b {
+		t.Errorf("Assert failed, expected %v, got %v", b, a)
+	}
+}
+
+func TestHumanDuration(t *testing.T) {
+	day := 24 * time.Hour
+	week := 7 * day
+	month := 30 * day
+	year := 365 * day
+
+	assertEqual(t, "Less than a second", HumanDuration(450*time.Millisecond))
+	assertEqual(t, "Less than a second", HumanDurationWithCase(450*time.Millisecond, true))
+	assertEqual(t, "less than a second", HumanDurationWithCase(450*time.Millisecond, false))
+	assertEqual(t, "1 second", HumanDuration(1*time.Second))
+	assertEqual(t, "45 seconds", HumanDuration(45*time.Second))
+	assertEqual(t, "46 seconds", HumanDuration(46*time.Second))
+	assertEqual(t, "59 seconds", HumanDuration(59*time.Second))
+	assertEqual(t, "About a minute", HumanDuration(60*time.Second))
+	assertEqual(t, "About a minute", HumanDurationWithCase(1*time.Minute, true))
+	assertEqual(t, "about a minute", HumanDurationWithCase(1*time.Minute, false))
+	assertEqual(t, "3 minutes", HumanDuration(3*time.Minute))
+	assertEqual(t, "35 minutes", HumanDuration(35*time.Minute))
+	assertEqual(t, "35 minutes", HumanDuration(35*time.Minute+40*time.Second))
+	assertEqual(t, "45 minutes", HumanDuration(45*time.Minute))
+	assertEqual(t, "45 minutes", HumanDuration(45*time.Minute+40*time.Second))
+	assertEqual(t, "46 minutes", HumanDuration(46*time.Minute))
+	assertEqual(t, "59 minutes", HumanDuration(59*time.Minute))
+	assertEqual(t, "About an hour", HumanDuration(1*time.Hour))
+	assertEqual(t, "About an hour", HumanDurationWithCase(1*time.Hour+29*time.Minute, true))
+	assertEqual(t, "about an hour", HumanDurationWithCase(1*time.Hour+29*time.Minute, false))
+	assertEqual(t, "2 hours", HumanDuration(1*time.Hour+31*time.Minute))
+	assertEqual(t, "2 hours", HumanDuration(1*time.Hour+59*time.Minute))
+	assertEqual(t, "3 hours", HumanDuration(3*time.Hour))
+	assertEqual(t, "3 hours", HumanDuration(3*time.Hour+29*time.Minute))
+	assertEqual(t, "4 hours", HumanDuration(3*time.Hour+31*time.Minute))
+	assertEqual(t, "4 hours", HumanDuration(3*time.Hour+59*time.Minute))
+	assertEqual(t, "4 hours", HumanDuration(3*time.Hour+60*time.Minute))
+	assertEqual(t, "24 hours", HumanDuration(24*time.Hour))
+	assertEqual(t, "36 hours", HumanDuration(1*day+12*time.Hour))
+	assertEqual(t, "2 days", HumanDuration(2*day))
+	assertEqual(t, "7 days", HumanDuration(7*day))
+	assertEqual(t, "13 days", HumanDuration(13*day+5*time.Hour))
+	assertEqual(t, "2 weeks", HumanDuration(2*week))
+	assertEqual(t, "2 weeks", HumanDuration(2*week+4*day))
+	assertEqual(t, "3 weeks", HumanDuration(3*week))
+	assertEqual(t, "4 weeks", HumanDuration(4*week))
+	assertEqual(t, "4 weeks", HumanDuration(4*week+3*day))
+	assertEqual(t, "4 weeks", HumanDuration(1*month))
+	assertEqual(t, "6 weeks", HumanDuration(1*month+2*week))
+	assertEqual(t, "2 months", HumanDuration(2*month))
+	assertEqual(t, "2 months", HumanDuration(2*month+2*week))
+	assertEqual(t, "3 months", HumanDuration(3*month))
+	assertEqual(t, "3 months", HumanDuration(3*month+1*week))
+	assertEqual(t, "5 months", HumanDuration(5*month+2*week))
+	assertEqual(t, "13 months", HumanDuration(13*month))
+	assertEqual(t, "23 months", HumanDuration(23*month))
+	assertEqual(t, "24 months", HumanDuration(24*month))
+	assertEqual(t, "2 years", HumanDuration(24*month+2*week))
+	assertEqual(t, "3 years", HumanDuration(3*year+2*month))
+}
+
+func TestHumanTime(t *testing.T) {
+	now := time.Now()
+
+	t.Run("zero value", func(t *testing.T) {
+		assertEqual(t, HumanTime(time.Time{}, "never"), "never")
+	})
+	t.Run("time in the future", func(t *testing.T) {
+		v := now.Add(48 * time.Hour)
+		assertEqual(t, HumanTime(v, ""), "2 days from now")
+	})
+	t.Run("time in the past", func(t *testing.T) {
+		v := now.Add(-48 * time.Hour)
+		assertEqual(t, HumanTime(v, ""), "2 days ago")
+	})
+}
+
+func TestExactDuration(t *testing.T) {
+	assertEqual(t, "1 millisecond", ExactDuration(1*time.Millisecond))
+	assertEqual(t, "10 milliseconds", ExactDuration(10*time.Millisecond))
+	assertEqual(t, "1 second", ExactDuration(1*time.Second))
+	assertEqual(t, "10 seconds", ExactDuration(10*time.Second))
+	assertEqual(t, "1 minute", ExactDuration(1*time.Minute))
+	assertEqual(t, "10 minutes", ExactDuration(10*time.Minute))
+	assertEqual(t, "1 hour", ExactDuration(1*time.Hour))
+	assertEqual(t, "10 hours", ExactDuration(10*time.Hour))
+	assertEqual(t, "1 hour 1 second", ExactDuration(1*time.Hour+1*time.Second))
+	assertEqual(t, "1 hour 10 seconds", ExactDuration(1*time.Hour+10*time.Second))
+	assertEqual(t, "1 hour 1 minute", ExactDuration(1*time.Hour+1*time.Minute))
+	assertEqual(t, "1 hour 10 minutes", ExactDuration(1*time.Hour+10*time.Minute))
+	assertEqual(t, "1 hour 1 minute 1 second", ExactDuration(1*time.Hour+1*time.Minute+1*time.Second))
+	assertEqual(t, "10 hours 10 minutes 10 seconds", ExactDuration(10*time.Hour+10*time.Minute+10*time.Second))
+}
--- a/go.mod
+++ b/go.mod
@@ -3,20 +3,22 @@ module github.com/jmorganca/ollama
 go 1.20

 require (
+	github.com/dustin/go-humanize v1.0.1
 	github.com/gin-gonic/gin v1.9.1
+	github.com/mattn/go-runewidth v0.0.14
+	github.com/mitchellh/colorstring v0.0.0-20190213212951-d06e56a500db
+	github.com/olekukonko/tablewriter v0.0.5
+	github.com/pdevine/readline v1.5.2
 	github.com/spf13/cobra v1.7.0
 )

-require (
-	github.com/mattn/go-runewidth v0.0.14 // indirect
-	github.com/mitchellh/colorstring v0.0.0-20190213212951-d06e56a500db // indirect
-	github.com/rivo/uniseg v0.2.0 // indirect
-)
+require github.com/rivo/uniseg v0.2.0 // indirect

 require (
 	github.com/bytedance/sonic v1.9.1 // indirect
 	github.com/chenzhuoyu/base64x v0.0.0-20221115062448-fe3a3abad311 // indirect
 	github.com/gabriel-vasile/mimetype v1.4.2 // indirect
+	github.com/gin-contrib/cors v1.4.0
 	github.com/gin-contrib/sse v0.1.0 // indirect
 	github.com/go-playground/locales v0.14.1 // indirect
 	github.com/go-playground/universal-translator v0.18.1 // indirect
@@ -27,21 +29,22 @@ require (
 	github.com/json-iterator/go v1.1.12 // indirect
 	github.com/klauspost/cpuid/v2 v2.2.4 // indirect
 	github.com/leodido/go-urn v1.2.4 // indirect
-	github.com/lithammer/fuzzysearch v1.1.8
 	github.com/mattn/go-isatty v0.0.19 // indirect
 	github.com/modern-go/concurrent v0.0.0-20180306012644-bacd9c7ef1dd // indirect
 	github.com/modern-go/reflect2 v1.0.2 // indirect
+	github.com/pbnjay/memory v0.0.0-20210728143218-7b4eea64cf58
 	github.com/pelletier/go-toml/v2 v2.0.8 // indirect
-	github.com/schollz/progressbar/v3 v3.13.1
 	github.com/spf13/pflag v1.0.5 // indirect
 	github.com/twitchyliquid64/golang-asm v0.15.1 // indirect
 	github.com/ugorji/go/codec v1.2.11 // indirect
 	golang.org/x/arch v0.3.0 // indirect
-	golang.org/x/crypto v0.10.0 // indirect
+	golang.org/x/crypto v0.10.0
+	golang.org/x/exp v0.0.0-20230817173708-d852ddb80c63
 	golang.org/x/net v0.10.0 // indirect
-	golang.org/x/sys v0.10.0 // indirect
+	golang.org/x/sys v0.11.0 // indirect
 	golang.org/x/term v0.10.0
 	golang.org/x/text v0.10.0 // indirect
+	gonum.org/v1/gonum v0.13.0
 	google.golang.org/protobuf v1.30.0 // indirect
 	gopkg.in/yaml.v3 v3.0.1 // indirect
 )
--- a/go.sum
+++ b/go.sum
@@ -4,23 +4,38 @@ github.com/bytedance/sonic v1.9.1/go.mod h1:i736AoUSYt75HyZLoJW9ERYxcy6eaN6h4BZX
 github.com/chenzhuoyu/base64x v0.0.0-20211019084208-fb5309c8db06/go.mod h1:DH46F32mSOjUmXrMHnKwZdA8wcEefY7UVqBKYGjpdQY=
 github.com/chenzhuoyu/base64x v0.0.0-20221115062448-fe3a3abad311 h1:qSGYFH7+jGhDF8vLC+iwCD4WpbV1EBDSzWkJODFLams=
 github.com/chenzhuoyu/base64x v0.0.0-20221115062448-fe3a3abad311/go.mod h1:b583jCggY9gE99b6G5LEC39OIiVsWj+R97kbl5odCEk=
+github.com/chzyer/logex v1.2.1 h1:XHDu3E6q+gdHgsdTPH6ImJMIp436vR6MPtH8gP05QzM=
+github.com/chzyer/logex v1.2.1/go.mod h1:JLbx6lG2kDbNRFnfkgvh4eRJRPX1QCoOIWomwysCBrQ=
+github.com/chzyer/test v1.0.0 h1:p3BQDXSxOhOG0P9z6/hGnII4LGiEPOYBhs8asl/fC04=
+github.com/chzyer/test v1.0.0/go.mod h1:2JlltgoNkt4TW/z9V/IzDdFaMTM2JPIi26O1pF38GC8=
 github.com/cpuguy83/go-md2man/v2 v2.0.2/go.mod h1:tgQtvFlXSQOSOSIRvRPT7W67SCa46tRHOmNcaadrF8o=
+github.com/creack/pty v1.1.9/go.mod h1:oKZEueFk5CKHvIhNR5MUki03XCEU+Q6VDXinZuGJ33E=
 github.com/davecgh/go-spew v1.1.0/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
 github.com/davecgh/go-spew v1.1.1 h1:vj9j/u1bqnvCEfJOwUhtlOARqs3+rkHYY13jYWTU97c=
 github.com/davecgh/go-spew v1.1.1/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
+github.com/dustin/go-humanize v1.0.1 h1:GzkhY7T5VNhEkwH0PVJgjz+fX1rhBrR7pRT3mDkpeCY=
+github.com/dustin/go-humanize v1.0.1/go.mod h1:Mu1zIs6XwVuF/gI1OepvI0qD18qycQx+mFykh5fBlto=
 github.com/gabriel-vasile/mimetype v1.4.2 h1:w5qFW6JKBz9Y393Y4q372O9A7cUSequkh1Q7OhCmWKU=
 github.com/gabriel-vasile/mimetype v1.4.2/go.mod h1:zApsH/mKG4w07erKIaJPFiX0Tsq9BFQgN3qGY5GnNgA=
+github.com/gin-contrib/cors v1.4.0 h1:oJ6gwtUl3lqV0WEIwM/LxPF1QZ5qe2lGWdY2+bz7y0g=
+github.com/gin-contrib/cors v1.4.0/go.mod h1:bs9pNM0x/UsmHPBWT2xZz9ROh8xYjYkiURUfmBoMlcs=
 github.com/gin-contrib/sse v0.1.0 h1:Y/yl/+YNO8GZSjAhjMsSuLt29uWRFHdHYUb5lYOV9qE=
 github.com/gin-contrib/sse v0.1.0/go.mod h1:RHrZQHXnP2xjPF+u1gW/2HnVO7nvIa9PG3Gm+fLHvGI=
+github.com/gin-gonic/gin v1.8.1/go.mod h1:ji8BvRH1azfM+SYow9zQ6SZMvR8qOMZHmsCuWR9tTTk=
 github.com/gin-gonic/gin v1.9.1 h1:4idEAncQnU5cB7BeOkPtxjfCSye0AAm1R0RVIqJ+Jmg=
 github.com/gin-gonic/gin v1.9.1/go.mod h1:hPrL7YrpYKXt5YId3A/Tnip5kqbEAP+KLuI3SUcPTeU=
+github.com/go-playground/assert/v2 v2.0.1/go.mod h1:VDjEfimB/XKnb+ZQfWdccd7VUvScMdVu0Titje2rxJ4=
 github.com/go-playground/assert/v2 v2.2.0 h1:JvknZsQTYeFEAhQwI4qEt9cyV5ONwRHC+lYKSsYSR8s=
+github.com/go-playground/locales v0.14.0/go.mod h1:sawfccIbzZTqEDETgFXqTho0QybSa7l++s0DH+LDiLs=
 github.com/go-playground/locales v0.14.1 h1:EWaQ/wswjilfKLTECiXz7Rh+3BjFhfDFKv/oXslEjJA=
 github.com/go-playground/locales v0.14.1/go.mod h1:hxrqLVvrK65+Rwrd5Fc6F2O76J/NuW9t0sjnWqG1slY=
+github.com/go-playground/universal-translator v0.18.0/go.mod h1:UvRDBj+xPUEGrFYl+lu/H90nyDXpg0fqeB/AQUGNTVA=
 github.com/go-playground/universal-translator v0.18.1 h1:Bcnm0ZwsGyWbCzImXv+pAJnYK9S473LQFuzCbDbfSFY=
 github.com/go-playground/universal-translator v0.18.1/go.mod h1:xekY+UJKNuX9WP91TpwSH2VMlDf28Uj24BCp08ZFTUY=
+github.com/go-playground/validator/v10 v10.10.0/go.mod h1:74x4gJWsvQexRdW8Pn3dXSGrTK4nAUsbPlLADvpJkos=
 github.com/go-playground/validator/v10 v10.14.0 h1:vgvQWe3XCz3gIeFDm/HnTIbj6UGmg/+t63MyGU2n5js=
 github.com/go-playground/validator/v10 v10.14.0/go.mod h1:9iXMNT7sEkjXb0I+enO7QXmzG6QCsPWY4zveKFVRSyU=
+github.com/goccy/go-json v0.9.7/go.mod h1:6MelG93GURQebXPDq3khkgXZkazVtN9CRI+MGFi0w8I=
 github.com/goccy/go-json v0.10.2 h1:CrxCmQqYDkv1z7lO7Wbh2HN93uovUHgrECaO5ZrCXAU=
 github.com/goccy/go-json v0.10.2/go.mod h1:6MelG93GURQebXPDq3khkgXZkazVtN9CRI+MGFi0w8I=
 github.com/golang/protobuf v1.5.0/go.mod h1:FsONVRAS9T7sI+LIUmWTfcYkHO4aIWwzhcaSAoJOfIk=
@@ -32,17 +47,24 @@ github.com/inconshreveable/mousetrap v1.1.0 h1:wN+x4NVGpMsO7ErUn/mUI3vEoE6Jt13X2
 github.com/inconshreveable/mousetrap v1.1.0/go.mod h1:vpF70FUmC8bwa3OWnCshd2FqLfsEA9PFc4w1p2J65bw=
 github.com/json-iterator/go v1.1.12 h1:PV8peI4a0ysnczrg+LtxykD8LfKY9ML6u2jnxaEnrnM=
 github.com/json-iterator/go v1.1.12/go.mod h1:e30LSqwooZae/UwlEbR2852Gd8hjQvJoHmT4TnhNGBo=
-github.com/k0kubun/go-ansi v0.0.0-20180517002512-3bf9e2903213/go.mod h1:vNUNkEQ1e29fT/6vq2aBdFsgNPmy8qMdSay1npru+Sw=
 github.com/klauspost/cpuid/v2 v2.0.9/go.mod h1:FInQzS24/EEf25PyTYn52gqo7WaD8xa0213Md/qVLRg=
 github.com/klauspost/cpuid/v2 v2.2.4 h1:acbojRNwl3o09bUq+yDCtZFc1aiwaAAxtcn8YkZXnvk=
 github.com/klauspost/cpuid/v2 v2.2.4/go.mod h1:RVVoqg1df56z8g3pUjL/3lE5UfnlrJX8tyFgg4nqhuY=
+github.com/kr/pretty v0.1.0/go.mod h1:dAy3ld7l9f0ibDNOQOHHMYYIIbhfbHSm3C4ZsoJORNo=
+github.com/kr/pretty v0.2.1/go.mod h1:ipq/a2n7PKx3OHsz4KJII5eveXtPO4qwEXGdVfWzfnI=
+github.com/kr/pretty v0.3.0 h1:WgNl7dwNpEZ6jJ9k1snq4pZsg7DOEN8hP9Xw0Tsjwk0=
+github.com/kr/pretty v0.3.0/go.mod h1:640gp4NfQd8pI5XOwp5fnNeVWj67G7CFk/SaSQn7NBk=
+github.com/kr/pty v1.1.1/go.mod h1:pFQYn66WHrOpPYNljwOMqo10TkYh1fy3cYio2l3bCsQ=
+github.com/kr/text v0.1.0/go.mod h1:4Jbv+DJW3UT/LiOwJeYQe1efqtUx/iVham/4vfdArNI=
+github.com/kr/text v0.2.0 h1:5Nx0Ya0ZqY2ygV366QzturHI13Jq95ApcVaJBhpS+AY=
+github.com/kr/text v0.2.0/go.mod h1:eLer722TekiGuMkidMxC/pM04lWEeraHUUmBw8l2grE=
+github.com/leodido/go-urn v1.2.1/go.mod h1:zt4jvISO2HfUBqxjfIshjdMTYS56ZS/qv49ictyFfxY=
 github.com/leodido/go-urn v1.2.4 h1:XlAE/cm/ms7TE/VMVoduSpNBoyc2dOxHs5MZSwAN63Q=
 github.com/leodido/go-urn v1.2.4/go.mod h1:7ZrI8mTSeBSHl/UaRyKQW1qZeMgak41ANeCNaVckg+4=
-github.com/lithammer/fuzzysearch v1.1.8 h1:/HIuJnjHuXS8bKaiTMeeDlW2/AyIWk2brx1V8LFgLN4=
-github.com/lithammer/fuzzysearch v1.1.8/go.mod h1:IdqeyBClc3FFqSzYq/MXESsS4S0FsZ5ajtkr5xPLts4=
-github.com/mattn/go-isatty v0.0.17/go.mod h1:kYGgaQfpe5nmfYZH+SKPsOc2e4SrIfOl2e/yFXSvRLM=
+github.com/mattn/go-isatty v0.0.14/go.mod h1:7GGIvUiUoEMVVmxf/4nioHXj79iQHKdU27kJ6hsGG94=
 github.com/mattn/go-isatty v0.0.19 h1:JITubQf0MOLdlGRuRq+jtsDlekdYPia9ZFsB8h/APPA=
 github.com/mattn/go-isatty v0.0.19/go.mod h1:W+V8PltTTMOvKvAeJH7IuucS94S2C6jfK/D7dTCTo3Y=
+github.com/mattn/go-runewidth v0.0.9/go.mod h1:H031xJmbD/WCDINGzjvQ9THkh0rPKHF+m2gUSrubnMI=
 github.com/mattn/go-runewidth v0.0.14 h1:+xnbZSEeDbOIg5/mE6JF0w6n9duR1l3/WmbinWVwUuU=
 github.com/mattn/go-runewidth v0.0.14/go.mod h1:Jdepj2loyihRzMpdS35Xk/zdY8IAYHsh153qUoGf23w=
 github.com/mitchellh/colorstring v0.0.0-20190213212951-d06e56a500db h1:62I3jR2EmQ4l5rM/4FEfDWcRD+abF5XlKShorW5LRoQ=
@@ -52,15 +74,24 @@ github.com/modern-go/concurrent v0.0.0-20180306012644-bacd9c7ef1dd h1:TRLaZ9cD/w
 github.com/modern-go/concurrent v0.0.0-20180306012644-bacd9c7ef1dd/go.mod h1:6dJC0mAP4ikYIbvyc7fijjWJddQyLn8Ig3JB5CqoB9Q=
 github.com/modern-go/reflect2 v1.0.2 h1:xBagoLtFs94CBntxluKeaWgTMpvLxC4ur3nMaC9Gz0M=
 github.com/modern-go/reflect2 v1.0.2/go.mod h1:yWuevngMOJpCy52FWWMvUC8ws7m/LJsjYzDa0/r8luk=
+github.com/olekukonko/tablewriter v0.0.5 h1:P2Ga83D34wi1o9J6Wh1mRuqd4mF/x/lgBS7N7AbDhec=
+github.com/olekukonko/tablewriter v0.0.5/go.mod h1:hPp6KlRPjbx+hW8ykQs1w3UBbZlj6HuIJcUGPhkA7kY=
+github.com/pbnjay/memory v0.0.0-20210728143218-7b4eea64cf58 h1:onHthvaw9LFnH4t2DcNVpwGmV9E1BkGknEliJkfwQj0=
+github.com/pbnjay/memory v0.0.0-20210728143218-7b4eea64cf58/go.mod h1:DXv8WO4yhMYhSNPKjeNKa5WY9YCIEBRbNzFFPJbWO6Y=
+github.com/pdevine/readline v1.5.2 h1:oz6Y5GdTmhPG+08hhxcAvtHitSANWuA2100Sppb38xI=
+github.com/pdevine/readline v1.5.2/go.mod h1:na/LbuE5PYwxI7GyopWdIs3U8HVe89lYlNTFTXH3wOw=
+github.com/pelletier/go-toml/v2 v2.0.1/go.mod h1:r9LEWfGN8R5k0VXJ+0BkIe7MYkRdwZOjgMj2KwnJFUo=
 github.com/pelletier/go-toml/v2 v2.0.8 h1:0ctb6s9mE31h0/lhu+J6OPmVeDxJn+kYnJc2jZR9tGQ=
 github.com/pelletier/go-toml/v2 v2.0.8/go.mod h1:vuYfssBdrU2XDZ9bYydBu6t+6a6PYNcZljzZR9VXg+4=
+github.com/pkg/diff v0.0.0-20210226163009-20ebb0f2a09e/go.mod h1:pJLUxLENpZxwdsKMEsNbx1VGcRFpLqf3715MtcvvzbA=
 github.com/pmezard/go-difflib v1.0.0 h1:4DBwDE0NGyQoBHbLQYPwSUPoCMWR5BEzIk/f1lZbAQM=
 github.com/pmezard/go-difflib v1.0.0/go.mod h1:iKH77koFhYxTK1pcRnkKkqfTogsbg7gZNVY4sRDYZ/4=
 github.com/rivo/uniseg v0.2.0 h1:S1pD9weZBuJdFmowNwbpi7BJ8TNftyUImj/0WQi72jY=
 github.com/rivo/uniseg v0.2.0/go.mod h1:J6wj4VEh+S6ZtnVlnTBMWIodfgj8LQOQFoIToxlJtxc=
+github.com/rogpeppe/go-internal v1.6.1/go.mod h1:xXDCJY+GAPziupqXw64V24skbSoqbTEfhy4qGm1nDQc=
+github.com/rogpeppe/go-internal v1.8.0 h1:FCbCCtXNOY3UtUuHUYaghJg4y7Fd14rXifAYUAtL9R8=
+github.com/rogpeppe/go-internal v1.8.0/go.mod h1:WmiCO8CzOY8rg0OYDC4/i/2WRWAB6poM+XZ2dLUbcbE=
 github.com/russross/blackfriday/v2 v2.1.0/go.mod h1:+Rmxgy9KzJVeS9/2gXHxylqXiyQDYRxCVz55jmeOWTM=
-github.com/schollz/progressbar/v3 v3.13.1 h1:o8rySDYiQ59Mwzy2FELeHY5ZARXZTVJC7iHD6PEFUiE=
-github.com/schollz/progressbar/v3 v3.13.1/go.mod h1:xvrbki8kfT1fzWzBT/UZd9L6GA+jdL7HAgq2RFnO6fQ=
 github.com/spf13/cobra v1.7.0 h1:hyqWnYt1ZQShIddO5kBpj3vu05/++x6tJ6dg8EC572I=
 github.com/spf13/cobra v1.7.0/go.mod h1:uLxZILRyS/50WlhOIKD7W6V5bgeIt+4sICxh6uRMrb0=
 github.com/spf13/pflag v1.0.5 h1:iy+VFUOCP1a+8yFto/drg2CJ5u0yRoB7fZw3DKv/JXA=
@@ -69,6 +100,7 @@ github.com/stretchr/objx v0.1.0/go.mod h1:HFkY916IF+rwdDfMAkV7OtwuqBVzrE8GR6GFx+
 github.com/stretchr/objx v0.4.0/go.mod h1:YvHI0jy2hoMjB+UWwv71VJQ9isScKT/TqJzVSSt89Yw=
 github.com/stretchr/objx v0.5.0/go.mod h1:Yh+to48EsGEfYuaHDzXPcE3xhTkx73EhmCGUpEOglKo=
 github.com/stretchr/testify v1.3.0/go.mod h1:M5WIy9Dh21IEIfnGCwXGc5bZfKNJtfHm1UVUgZn+9EI=
+github.com/stretchr/testify v1.6.1/go.mod h1:6Fq8oRcR53rry900zMqJjRRixrwX3KX962/h/Wwjteg=
 github.com/stretchr/testify v1.7.0/go.mod h1:6Fq8oRcR53rry900zMqJjRRixrwX3KX962/h/Wwjteg=
 github.com/stretchr/testify v1.7.1/go.mod h1:6Fq8oRcR53rry900zMqJjRRixrwX3KX962/h/Wwjteg=
 github.com/stretchr/testify v1.8.0/go.mod h1:yNjHg4UonilssWZ8iaSj1OCr/vHnekPRkoO+kdMU+MU=
@@ -78,63 +110,53 @@ github.com/stretchr/testify v1.8.3 h1:RP3t2pwF7cMEbC1dqtB6poj3niw/9gnV4Cjg5oW5gt
 github.com/stretchr/testify v1.8.3/go.mod h1:sz/lmYIOXD/1dqDmKjjqLyZ2RngseejIcXlSw2iwfAo=
 github.com/twitchyliquid64/golang-asm v0.15.1 h1:SU5vSMR7hnwNxj24w34ZyCi/FmDZTkS4MhqMhdFk5YI=
 github.com/twitchyliquid64/golang-asm v0.15.1/go.mod h1:a1lVb/DtPvCB8fslRZhAngC2+aY1QWCk3Cedj/Gdt08=
+github.com/ugorji/go v1.2.7/go.mod h1:nF9osbDWLy6bDVv/Rtoh6QgnvNDpmCalQV5urGCCS6M=
+github.com/ugorji/go/codec v1.2.7/go.mod h1:WGN1fab3R1fzQlVQTkfxVtIBhWDRqOviHU95kRgeqEY=
 github.com/ugorji/go/codec v1.2.11 h1:BMaWp1Bb6fHwEtbplGBGJ498wD+LKlNSl25MjdZY4dU=
 github.com/ugorji/go/codec v1.2.11/go.mod h1:UNopzCgEMSXjBc6AOMqYvWC1ktqTAfzJZUZgYf6w6lg=
-github.com/yuin/goldmark v1.4.13/go.mod h1:6yULJ656Px+3vBD8DxQVa3kxgyrAnzto9xy5taEt/CY=
 golang.org/x/arch v0.0.0-20210923205945-b76863e36670/go.mod h1:5om86z9Hs0C8fWVUuoMHwpExlXzs5Tkyp9hOrfG7pp8=
 golang.org/x/arch v0.3.0 h1:02VY4/ZcO/gBOH6PUaoiptASxtXU10jazRCP865E97k=
 golang.org/x/arch v0.3.0/go.mod h1:5om86z9Hs0C8fWVUuoMHwpExlXzs5Tkyp9hOrfG7pp8=
-golang.org/x/crypto v0.0.0-20190308221718-c2843e01d9a2/go.mod h1:djNgcEr1/C05ACkg1iLfiJU5Ep61QUkGW8qpdssI0+w=
-golang.org/x/crypto v0.0.0-20210921155107-089bfa567519/go.mod h1:GvvjBRRGRdwPK5ydBHafDWAxML/pGHZbMvKqRZ5+Abc=
+golang.org/x/crypto v0.0.0-20210711020723-a769d52b0f97/go.mod h1:GvvjBRRGRdwPK5ydBHafDWAxML/pGHZbMvKqRZ5+Abc=
 golang.org/x/crypto v0.10.0 h1:LKqV2xt9+kDzSTfOhx4FrkEBcMrAgHSYgzywV9zcGmM=
 golang.org/x/crypto v0.10.0/go.mod h1:o4eNf7Ede1fv+hwOwZsTHl9EsPFO6q6ZvYR8vYfY45I=
-golang.org/x/mod v0.6.0-dev.0.20220419223038-86c51ed26bb4/go.mod h1:jJ57K6gSWd91VN4djpZkiMVwK6gcyfeH4XE8wZrZaV4=
-golang.org/x/mod v0.8.0/go.mod h1:iBbtSCu2XBx23ZKBPSOrRkjjQPZFPuis4dIYUhu/chs=
-golang.org/x/net v0.0.0-20190620200207-3b0461eec859/go.mod h1:z5CRVTTTmAJ677TzLLGU+0bjPO0LkuOLi4/5GtJWs/s=
+golang.org/x/exp v0.0.0-20230817173708-d852ddb80c63 h1:m64FZMko/V45gv0bNmrNYoDEq8U5YUhetc9cBWKS1TQ=
+golang.org/x/exp v0.0.0-20230817173708-d852ddb80c63/go.mod h1:0v4NqG35kSWCMzLaMeX+IQrlSnVE/bqGSyC2cz/9Le8=
 golang.org/x/net v0.0.0-20210226172049-e18ecbb05110/go.mod h1:m0MpNAwzfU5UDzcl9v0D8zg8gWTRqZa9RBIspLL5mdg=
-golang.org/x/net v0.0.0-20220722155237-a158d28d115b/go.mod h1:XRhObCWvk6IyKnWLug+ECip1KBveYUHfp+8e9klMJ9c=
-golang.org/x/net v0.6.0/go.mod h1:2Tu9+aMcznHK/AK1HMvgo6xiTLG5rD5rZLDS+rp2Bjs=
 golang.org/x/net v0.10.0 h1:X2//UzNDwYmtCLn7To6G58Wr6f5ahEAQgKNzv9Y951M=
 golang.org/x/net v0.10.0/go.mod h1:0qNGK6F8kojg2nk9dLZ2mShWaEBan6FAoqfSigmmuDg=
-golang.org/x/sync v0.0.0-20190423024810-112230192c58/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
-golang.org/x/sync v0.0.0-20220722155255-886fb9371eb4/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
-golang.org/x/sync v0.1.0/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
-golang.org/x/sys v0.0.0-20190215142949-d0b11bdaac8a/go.mod h1:STP8DvDyc/dI5b8T5hshtkjS+E42TnysNCUPdjciGhY=
 golang.org/x/sys v0.0.0-20201119102817-f84b799fce68/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
 golang.org/x/sys v0.0.0-20210615035016-665e8c7367d1/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
-golang.org/x/sys v0.0.0-20220520151302-bc2c85ada10a/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
+golang.org/x/sys v0.0.0-20210630005230-0f9fa26af87c/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
+golang.org/x/sys v0.0.0-20210806184541-e5e7981a1069/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
+golang.org/x/sys v0.0.0-20220310020820-b874c991c1a5/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
 golang.org/x/sys v0.0.0-20220704084225-05e143d24a9e/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
-golang.org/x/sys v0.0.0-20220722155257-8c9f86f7a55f/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
-golang.org/x/sys v0.0.0-20220811171246-fbc7d0a398ab/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
-golang.org/x/sys v0.5.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
 golang.org/x/sys v0.6.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
-golang.org/x/sys v0.10.0 h1:SqMFp9UcQJZa+pmYuAKjd9xq1f0j5rLcDIk0mj4qAsA=
-golang.org/x/sys v0.10.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
+golang.org/x/sys v0.11.0 h1:eG7RXZHdqOJ1i+0lgLgCpSXAp6M3LYlAo6osgSi0xOM=
+golang.org/x/sys v0.11.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
 golang.org/x/term v0.0.0-20201126162022-7de9c90e9dd1/go.mod h1:bj7SfCRtBDWHUb9snDiAeCFNEtKQo2Wmx5Cou7ajbmo=
-golang.org/x/term v0.0.0-20210927222741-03fcf44c2211/go.mod h1:jbD1KX2456YbFQfuXm/mYQcufACuNUgVhRMnK/tPxf8=
-golang.org/x/term v0.5.0/go.mod h1:jMB1sMXY+tzblOD4FWmEbocvup2/aLOaQEp7JmGp78k=
-golang.org/x/term v0.6.0/go.mod h1:m6U89DPEgQRMq3DNkDClhWw02AUbt2daBVO4cn4Hv9U=
 golang.org/x/term v0.10.0 h1:3R7pNqamzBraeqj/Tj8qt1aQ2HpmlC+Cx/qL/7hn4/c=
 golang.org/x/term v0.10.0/go.mod h1:lpqdcUyK/oCiQxvxVrppt5ggO2KCZ5QblwqPnfZ6d5o=
-golang.org/x/text v0.3.0/go.mod h1:NqM8EUOU14njkJ3fqMW+pc6Ldnwhi/IjpwHt7yyuwOQ=
 golang.org/x/text v0.3.3/go.mod h1:5Zoc/QRtKVWzQhOtBMvqHzDpF6irO9z98xDceosuGiQ=
-golang.org/x/text v0.3.7/go.mod h1:u+2+/6zg+i71rQMx5EYifcz6MCKuco9NR6JIITiCfzQ=
-golang.org/x/text v0.7.0/go.mod h1:mrYo+phRRbMaCq/xk9113O4dZlRixOauAjOtrjsXDZ8=
-golang.org/x/text v0.9.0/go.mod h1:e1OnstbJyHTd6l/uOt8jFFHp6TRDWZR/bV3emEE/zU8=
+golang.org/x/text v0.3.6/go.mod h1:5Zoc/QRtKVWzQhOtBMvqHzDpF6irO9z98xDceosuGiQ=
 golang.org/x/text v0.10.0 h1:UpjohKhiEgNc0CSauXmwYftY1+LlaC75SJwh0SgCX58=
 golang.org/x/text v0.10.0/go.mod h1:TvPlkZtksWOMsz7fbANvkp4WM8x/WCo/om8BMLbz+aE=
 golang.org/x/tools v0.0.0-20180917221912-90fa682c2a6e/go.mod h1:n7NCudcB/nEzxVGmLbDWY5pfWTLqBcC2KZ6jyYvM4mQ=
-golang.org/x/tools v0.0.0-20191119224855-298f0cb1881e/go.mod h1:b+2E5dAYhXwXZwtnZ6UAqBI28+e2cm9otk0dWdXHAEo=
-golang.org/x/tools v0.1.12/go.mod h1:hNGJHUnrk76NpqgfD5Aqm5Crs+Hm0VOH/i9J2+nxYbc=
-golang.org/x/tools v0.6.0/go.mod h1:Xwgl3UAJ/d3gWutnCtw505GrjyAbvKui8lOU390QaIU=
-golang.org/x/xerrors v0.0.0-20190717185122-a985d3407aa7/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0=
 golang.org/x/xerrors v0.0.0-20191204190536-9bdfabe68543/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0=
+gonum.org/v1/gonum v0.13.0 h1:a0T3bh+7fhRyqeNbiC3qVHYmkiQgit3wnNan/2c0HMM=
+gonum.org/v1/gonum v0.13.0/go.mod h1:/WPYRckkfWrhWefxyYTfrTtQR0KH4iyHNuzxqXAKyAU=
 google.golang.org/protobuf v1.26.0-rc.1/go.mod h1:jlhhOSvTdKEhbULTjvd4ARK9grFBp09yW+WbY/TyQbw=
+google.golang.org/protobuf v1.28.0/go.mod h1:HV8QOd/L58Z+nl8r43ehVNZIU/HEI6OcFqwMG9pJV4I=
 google.golang.org/protobuf v1.30.0 h1:kPPoIgf3TsEvrm0PFe15JQ+570QVxYzEvvHqChK+cng=
 google.golang.org/protobuf v1.30.0/go.mod h1:HV8QOd/L58Z+nl8r43ehVNZIU/HEI6OcFqwMG9pJV4I=
-gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405 h1:yhCVgyC4o1eVCa2tZl7eS0r+SDo693bJlVdllGtEeKM=
 gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0=
+gopkg.in/check.v1 v1.0.0-20180628173108-788fd7840127/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0=
+gopkg.in/check.v1 v1.0.0-20201130134442-10cb98267c6c h1:Hei/4ADfdWqJk1ZMxUNpqntNwaWcugrBjAiHlqqRiVk=
+gopkg.in/check.v1 v1.0.0-20201130134442-10cb98267c6c/go.mod h1:JHkPIbrfpd72SG/EVd6muEfDQjcINNoR0C8j2r3qZ4Q=
+gopkg.in/errgo.v2 v2.1.0/go.mod h1:hNsd1EY+bozCKY1Ytp96fpM3vjJbqLJn88ws8XvfDNI=
+gopkg.in/yaml.v2 v2.4.0/go.mod h1:RDklbk79AGWmwhnvt/jBztapEOGDOx6ZbXqjP6csGnQ=
 gopkg.in/yaml.v3 v3.0.0-20200313102051-9f266ea9e77c/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM=
+gopkg.in/yaml.v3 v3.0.0-20210107192922-496545a6307b/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM=
 gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA=
 gopkg.in/yaml.v3 v3.0.1/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM=
 rsc.io/pdf v0.1.1/go.mod h1:n8OzWcQ6Sp37PL01nO98y4iUCRdTGarVfzxY20ICaU4=
--- a/llama/.gitignore
+++ b/llama/.gitignore
@@ -1 +0,0 @@
-build
--- a/llama/CMakeLists.txt
+++ b/llama/CMakeLists.txt
@@ -1,23 +0,0 @@
-cmake_minimum_required(VERSION 3.12)
-project(binding)
-
-include(FetchContent)
-
-FetchContent_Declare(
-    llama_cpp
-    GIT_REPOSITORY https://github.com/ggerganov/llama.cpp.git
-    GIT_TAG        55dbb91
-)
-
-FetchContent_MakeAvailable(llama_cpp)
-
-add_library(binding ${CMAKE_CURRENT_SOURCE_DIR}/binding/binding.cpp ${llama_cpp_SOURCE_DIR}/examples/common.cpp)
-target_include_directories(binding PRIVATE ${llama_cpp_SOURCE_DIR}/examples)
-target_link_libraries(binding llama ggml_static)
-
-if (LLAMA_METAL)
-    configure_file(${llama_cpp_SOURCE_DIR}/ggml-metal.metal ${CMAKE_CURRENT_BINARY_DIR}/../../ggml-metal.metal COPYONLY)
-endif()
-
-add_custom_target(copy_libllama ALL COMMAND ${CMAKE_COMMAND} -E copy_if_different $<TARGET_FILE:llama> ${CMAKE_CURRENT_BINARY_DIR})
-add_custom_target(copy_libggml_static ALL COMMAND ${CMAKE_COMMAND} -E copy_if_different $<TARGET_FILE:ggml_static> ${CMAKE_CURRENT_BINARY_DIR})
--- a/llama/binding/binding.cpp
+++ b/llama/binding/binding.cpp
@@ -1,705 +0,0 @@
-// MIT License
-
-// Copyright (c) 2023 go-skynet authors
-
-// Permission is hereby granted, free of charge, to any person obtaining a copy
-// of this software and associated documentation files (the "Software"), to deal
-// in the Software without restriction, including without limitation the rights
-// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
-// copies of the Software, and to permit persons to whom the Software is
-// furnished to do so, subject to the following conditions:
-
-// The above copyright notice and this permission notice shall be included in all
-// copies or substantial portions of the Software.
-
-// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
-// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
-// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
-// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
-// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
-// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
-// SOFTWARE.
-
-#include "common.h"
-#include "llama.h"
-
-#include "binding.h"
-
-#include <cassert>
-#include <cinttypes>
-#include <cmath>
-#include <cstdio>
-#include <cstring>
-#include <fstream>
-#include <iostream>
-#include <regex>
-#include <sstream>
-#include <string>
-#include <vector>
-#if defined(__unix__) || (defined(__APPLE__) && defined(__MACH__))
-#include <signal.h>
-#include <unistd.h>
-#elif defined(_WIN32)
-#define WIN32_LEAN_AND_MEAN
-#define NOMINMAX
-#include <signal.h>
-#include <windows.h>
-#endif
-
-#if defined(__unix__) || (defined(__APPLE__) && defined(__MACH__)) || \
-    defined(_WIN32)
-void sigint_handler(int signo) {
-  if (signo == SIGINT) {
-    _exit(130);
-  }
-}
-#endif
-
-int get_embeddings(void *params_ptr, void *state_pr, float *res_embeddings) {
-  gpt_params *params_p = (gpt_params *)params_ptr;
-  llama_context *ctx = (llama_context *)state_pr;
-  gpt_params params = *params_p;
-
-  if (params.seed <= 0) {
-    params.seed = time(NULL);
-  }
-
-  std::mt19937 rng(params.seed);
-
-  llama_init_backend(params.numa);
-
-  int n_past = 0;
-
-  // Add a space in front of the first character to match OG llama tokenizer
-  // behavior
-  params.prompt.insert(0, 1, ' ');
-
-  // tokenize the prompt
-  auto embd_inp = ::llama_tokenize(ctx, params.prompt, true);
-
-  // determine newline token
-  auto llama_token_newline = ::llama_tokenize(ctx, "\n", false);
-
-  if (embd_inp.size() > 0) {
-    if (llama_eval(ctx, embd_inp.data(), embd_inp.size(), n_past,
-                   params.n_threads)) {
-      fprintf(stderr, "%s : failed to eval\n", __func__);
-      return 1;
-    }
-  }
-
-  const int n_embd = llama_n_embd(ctx);
-
-  const auto embeddings = llama_get_embeddings(ctx);
-
-  for (int i = 0; i < n_embd; i++) {
-    res_embeddings[i] = embeddings[i];
-  }
-
-  return 0;
-}
-
-int get_token_embeddings(void *params_ptr, void *state_pr, int *tokens,
-                         int tokenSize, float *res_embeddings) {
-  gpt_params *params_p = (gpt_params *)params_ptr;
-  llama_context *ctx = (llama_context *)state_pr;
-  gpt_params params = *params_p;
-
-  for (int i = 0; i < tokenSize; i++) {
-    auto token_str = llama_token_to_str(ctx, tokens[i]);
-    if (token_str == nullptr) {
-      continue;
-    }
-    std::vector<std::string> my_vector;
-    std::string str_token(token_str); // create a new std::string from the char*
-    params_p->prompt += str_token;
-  }
-
-  return get_embeddings(params_ptr, state_pr, res_embeddings);
-}
-
-int eval(void *params_ptr, void *state_pr, char *text) {
-  gpt_params *params_p = (gpt_params *)params_ptr;
-  llama_context *ctx = (llama_context *)state_pr;
-
-  auto n_past = 0;
-  auto last_n_tokens_data =
-      std::vector<llama_token>(params_p->repeat_last_n, 0);
-
-  auto tokens = std::vector<llama_token>(params_p->n_ctx);
-  auto n_prompt_tokens =
-      llama_tokenize(ctx, text, tokens.data(), tokens.size(), true);
-
-  if (n_prompt_tokens < 1) {
-    fprintf(stderr, "%s : failed to tokenize prompt\n", __func__);
-    return 1;
-  }
-
-  // evaluate prompt
-  return llama_eval(ctx, tokens.data(), n_prompt_tokens, n_past,
-                    params_p->n_threads);
-}
-
-int llama_predict(void *params_ptr, void *state_pr, char *result, bool debug) {
-  gpt_params *params_p = (gpt_params *)params_ptr;
-  llama_context *ctx = (llama_context *)state_pr;
-
-  gpt_params params = *params_p;
-
-  const int n_ctx = llama_n_ctx(ctx);
-
-  if (params.seed <= 0) {
-    params.seed = time(NULL);
-  }
-
-  std::mt19937 rng(params.seed);
-
-  std::string path_session = params.path_prompt_cache;
-  std::vector<llama_token> session_tokens;
-
-  if (!path_session.empty()) {
-    if (debug) {
-      fprintf(stderr, "%s: attempting to load saved session from '%s'\n",
-              __func__, path_session.c_str());
-    }
-    // fopen to check for existing session
-    FILE *fp = std::fopen(path_session.c_str(), "rb");
-    if (fp != NULL) {
-      std::fclose(fp);
-
-      session_tokens.resize(n_ctx);
-      size_t n_token_count_out = 0;
-      if (!llama_load_session_file(
-              ctx, path_session.c_str(), session_tokens.data(),
-              session_tokens.capacity(), &n_token_count_out)) {
-        fprintf(stderr, "%s: error: failed to load session file '%s'\n",
-                __func__, path_session.c_str());
-        return 1;
-      }
-      session_tokens.resize(n_token_count_out);
-      llama_set_rng_seed(ctx, params.seed);
-      if (debug) {
-        fprintf(stderr, "%s: loaded a session with prompt size of %d tokens\n",
-                __func__, (int)session_tokens.size());
-      }
-    } else {
-      if (debug) {
-        fprintf(stderr, "%s: session file does not exist, will create\n",
-                __func__);
-      }
-    }
-  }
-
-  std::vector<llama_token> embd_inp;
-  if (!params.prompt.empty() || session_tokens.empty()) {
-    // Add a space in front of the first character to match OG llama tokenizer
-    // behavior
-    params.prompt.insert(0, 1, ' ');
-
-    embd_inp = ::llama_tokenize(ctx, params.prompt, true);
-  } else {
-    embd_inp = session_tokens;
-  }
-
-  // debug message about similarity of saved session, if applicable
-  size_t n_matching_session_tokens = 0;
-  if (session_tokens.size()) {
-    for (llama_token id : session_tokens) {
-      if (n_matching_session_tokens >= embd_inp.size() ||
-          id != embd_inp[n_matching_session_tokens]) {
-        break;
-      }
-      n_matching_session_tokens++;
-    }
-    if (debug) {
-      if (params.prompt.empty() &&
-          n_matching_session_tokens == embd_inp.size()) {
-        fprintf(stderr, "%s: using full prompt from session file\n", __func__);
-      } else if (n_matching_session_tokens >= embd_inp.size()) {
-        fprintf(stderr, "%s: session file has exact match for prompt!\n",
-                __func__);
-      } else if (n_matching_session_tokens < (embd_inp.size() / 2)) {
-        fprintf(stderr,
-                "%s: warning: session file has low similarity to prompt (%zu / "
-                "%zu tokens); will mostly be reevaluated\n",
-                __func__, n_matching_session_tokens, embd_inp.size());
-      } else {
-        fprintf(stderr, "%s: session file matches %zu / %zu tokens of prompt\n",
-                __func__, n_matching_session_tokens, embd_inp.size());
-      }
-    }
-  }
-  // if we will use the cache for the full prompt without reaching the end of
-  // the cache, force reevaluation of the last token token to recalculate the
-  // cached logits
-  if (!embd_inp.empty() && n_matching_session_tokens == embd_inp.size() &&
-      session_tokens.size() > embd_inp.size()) {
-    session_tokens.resize(embd_inp.size() - 1);
-  }
-  // number of tokens to keep when resetting context
-  if (params.n_keep < 0 || params.n_keep > (int)embd_inp.size()) {
-    params.n_keep = (int)embd_inp.size();
-  }
-
-  // determine newline token
-  auto llama_token_newline = ::llama_tokenize(ctx, "\n", false);
-
-  // TODO: replace with ring-buffer
-  std::vector<llama_token> last_n_tokens(n_ctx);
-  std::fill(last_n_tokens.begin(), last_n_tokens.end(), 0);
-
-  bool need_to_save_session =
-      !path_session.empty() && n_matching_session_tokens < embd_inp.size();
-  int n_past = 0;
-  int n_remain = params.n_predict;
-  int n_consumed = 0;
-  int n_session_consumed = 0;
-
-  std::vector<llama_token> embd;
-  std::string res = "";
-
-  // do one empty run to warm up the model
-  {
-    const std::vector<llama_token> tmp = {
-        llama_token_bos(),
-    };
-    llama_eval(ctx, tmp.data(), tmp.size(), 0, params.n_threads);
-    llama_reset_timings(ctx);
-  }
-
-  while (n_remain != 0) {
-    // predict
-    if (embd.size() > 0) {
-      // infinite text generation via context swapping
-      // if we run out of context:
-      // - take the n_keep first tokens from the original prompt (via n_past)
-      // - take half of the last (n_ctx - n_keep) tokens and recompute the
-      // logits in batches
-      if (n_past + (int)embd.size() > n_ctx) {
-        const int n_left = n_past - params.n_keep;
-
-        // always keep the first token - BOS
-        n_past = std::max(1, params.n_keep);
-
-        // insert n_left/2 tokens at the start of embd from last_n_tokens
-        embd.insert(embd.begin(),
-                    last_n_tokens.begin() + n_ctx - n_left / 2 - embd.size(),
-                    last_n_tokens.end() - embd.size());
-
-        // stop saving session if we run out of context
-        path_session.clear();
-
-        // printf("\n---\n");
-        // printf("resetting: '");
-        // for (int i = 0; i < (int) embd.size(); i++) {
-        //     printf("%s", llama_token_to_str(ctx, embd[i]));
-        // }
-        // printf("'\n");
-        // printf("\n---\n");
-      }
-
-      // try to reuse a matching prefix from the loaded session instead of
-      // re-eval (via n_past)
-      if (n_session_consumed < (int)session_tokens.size()) {
-        size_t i = 0;
-        for (; i < embd.size(); i++) {
-          if (embd[i] != session_tokens[n_session_consumed]) {
-            session_tokens.resize(n_session_consumed);
-            break;
-          }
-
-          n_past++;
-          n_session_consumed++;
-
-          if (n_session_consumed >= (int)session_tokens.size()) {
-            ++i;
-            break;
-          }
-        }
-        if (i > 0) {
-          embd.erase(embd.begin(), embd.begin() + i);
-        }
-      }
-
-      // evaluate tokens in batches
-      // embd is typically prepared beforehand to fit within a batch, but not
-      // always
-      for (int i = 0; i < (int)embd.size(); i += params.n_batch) {
-        int n_eval = (int)embd.size() - i;
-        if (n_eval > params.n_batch) {
-          n_eval = params.n_batch;
-        }
-        if (llama_eval(ctx, &embd[i], n_eval, n_past, params.n_threads)) {
-          fprintf(stderr, "%s : failed to eval\n", __func__);
-          return 1;
-        }
-        n_past += n_eval;
-      }
-
-      if (embd.size() > 0 && !path_session.empty()) {
-        session_tokens.insert(session_tokens.end(), embd.begin(), embd.end());
-        n_session_consumed = session_tokens.size();
-      }
-    }
-
-    embd.clear();
-
-    if ((int)embd_inp.size() <= n_consumed) {
-      // out of user input, sample next token
-      const float temp = params.temp;
-      const int32_t top_k =
-          params.top_k <= 0 ? llama_n_vocab(ctx) : params.top_k;
-      const float top_p = params.top_p;
-      const float tfs_z = params.tfs_z;
-      const float typical_p = params.typical_p;
-      const int32_t repeat_last_n =
-          params.repeat_last_n < 0 ? n_ctx : params.repeat_last_n;
-      const float repeat_penalty = params.repeat_penalty;
-      const float alpha_presence = params.presence_penalty;
-      const float alpha_frequency = params.frequency_penalty;
-      const int mirostat = params.mirostat;
-      const float mirostat_tau = params.mirostat_tau;
-      const float mirostat_eta = params.mirostat_eta;
-      const bool penalize_nl = params.penalize_nl;
-
-      // optionally save the session on first sample (for faster prompt loading
-      // next time)
-      if (!path_session.empty() && need_to_save_session &&
-          !params.prompt_cache_ro) {
-        need_to_save_session = false;
-        llama_save_session_file(ctx, path_session.c_str(),
-                                session_tokens.data(), session_tokens.size());
-      }
-
-      llama_token id = 0;
-
-      {
-        auto logits = llama_get_logits(ctx);
-        auto n_vocab = llama_n_vocab(ctx);
-
-        // Apply params.logit_bias map
-        for (auto it = params.logit_bias.begin(); it != params.logit_bias.end();
-             it++) {
-          logits[it->first] += it->second;
-        }
-
-        std::vector<llama_token_data> candidates;
-        candidates.reserve(n_vocab);
-        for (llama_token token_id = 0; token_id < n_vocab; token_id++) {
-          candidates.emplace_back(
-              llama_token_data{token_id, logits[token_id], 0.0f});
-        }
-
-        llama_token_data_array candidates_p = {candidates.data(),
-                                               candidates.size(), false};
-
-        // Apply penalties
-        float nl_logit = logits[llama_token_nl()];
-        auto last_n_repeat =
-            std::min(std::min((int)last_n_tokens.size(), repeat_last_n), n_ctx);
-        llama_sample_repetition_penalty(
-            ctx, &candidates_p,
-            last_n_tokens.data() + last_n_tokens.size() - last_n_repeat,
-            last_n_repeat, repeat_penalty);
-        llama_sample_frequency_and_presence_penalties(
-            ctx, &candidates_p,
-            last_n_tokens.data() + last_n_tokens.size() - last_n_repeat,
-            last_n_repeat, alpha_frequency, alpha_presence);
-        if (!penalize_nl) {
-          logits[llama_token_nl()] = nl_logit;
-        }
-
-        if (temp <= 0) {
-          // Greedy sampling
-          id = llama_sample_token_greedy(ctx, &candidates_p);
-        } else {
-          if (mirostat == 1) {
-            static float mirostat_mu = 2.0f * mirostat_tau;
-            const int mirostat_m = 100;
-            llama_sample_temperature(ctx, &candidates_p, temp);
-            id = llama_sample_token_mirostat(ctx, &candidates_p, mirostat_tau,
-                                             mirostat_eta, mirostat_m,
-                                             &mirostat_mu);
-          } else if (mirostat == 2) {
-            static float mirostat_mu = 2.0f * mirostat_tau;
-            llama_sample_temperature(ctx, &candidates_p, temp);
-            id = llama_sample_token_mirostat_v2(
-                ctx, &candidates_p, mirostat_tau, mirostat_eta, &mirostat_mu);
-          } else {
-            // Temperature sampling
-            llama_sample_top_k(ctx, &candidates_p, top_k, 1);
-            llama_sample_tail_free(ctx, &candidates_p, tfs_z, 1);
-            llama_sample_typical(ctx, &candidates_p, typical_p, 1);
-            llama_sample_top_p(ctx, &candidates_p, top_p, 1);
-            llama_sample_temperature(ctx, &candidates_p, temp);
-            id = llama_sample_token(ctx, &candidates_p);
-          }
-        }
-        // printf("`%d`", candidates_p.size);
-
-        last_n_tokens.erase(last_n_tokens.begin());
-        last_n_tokens.push_back(id);
-      }
-
-      // add it to the context
-      embd.push_back(id);
-
-      // decrement remaining sampling budget
-      --n_remain;
-
-      // call the token callback, no need to check if one is actually
-      // registered, that will be handled on the Go side.
-      auto token_str = llama_token_to_str(ctx, id);
-      if (!tokenCallback(state_pr, (char *)token_str)) {
-        break;
-      }
-    } else {
-      // some user input remains from prompt or interaction, forward it to
-      // processing
-      while ((int)embd_inp.size() > n_consumed) {
-        embd.push_back(embd_inp[n_consumed]);
-        last_n_tokens.erase(last_n_tokens.begin());
-        last_n_tokens.push_back(embd_inp[n_consumed]);
-        ++n_consumed;
-        if ((int)embd.size() >= params.n_batch) {
-          break;
-        }
-      }
-    }
-
-    for (auto id : embd) {
-      res += llama_token_to_str(ctx, id);
-    }
-
-    // check for stop prompt
-    if (params.antiprompt.size()) {
-      std::string last_output;
-      for (auto id : last_n_tokens) {
-        last_output += llama_token_to_str(ctx, id);
-      }
-      // Check if each of the reverse prompts appears at the end of the output.
-      for (std::string &antiprompt : params.antiprompt) {
-        // size_t extra_padding = params.interactive ? 0 : 2;
-        size_t extra_padding = 2;
-        size_t search_start_pos =
-            last_output.length() >
-                    static_cast<size_t>(antiprompt.length() + extra_padding)
-                ? last_output.length() -
-                      static_cast<size_t>(antiprompt.length() + extra_padding)
-                : 0;
-
-        if (last_output.find(antiprompt.c_str(), search_start_pos) !=
-            std::string::npos) {
-          goto end;
-        }
-      }
-    }
-
-    // end of text token
-    if (!embd.empty() && embd.back() == llama_token_eos()) {
-      break;
-    }
-  }
-
-  if (!path_session.empty() && params.prompt_cache_all &&
-      !params.prompt_cache_ro) {
-    if (debug) {
-      fprintf(stderr, "\n%s: saving final output to session file '%s'\n",
-              __func__, path_session.c_str());
-    }
-    llama_save_session_file(ctx, path_session.c_str(), session_tokens.data(),
-                            session_tokens.size());
-  }
-
-end:
-#if defined(_WIN32)
-  signal(SIGINT, SIG_DFL);
-#endif
-
-  if (debug) {
-    llama_print_timings(ctx);
-    llama_reset_timings(ctx);
-  }
-
-  strcpy(result, res.c_str());
-  return 0;
-}
-
-void llama_binding_free_model(void *state_ptr) {
-  llama_context *ctx = (llama_context *)state_ptr;
-  llama_free(ctx);
-}
-
-void llama_free_params(void *params_ptr) {
-  gpt_params *params = (gpt_params *)params_ptr;
-  delete params;
-}
-
-int load_state(void *ctx, char *statefile, char *modes) {
-  llama_context *state = (llama_context *)ctx;
-  const llama_context *constState = static_cast<const llama_context *>(state);
-  const size_t state_size = llama_get_state_size(state);
-  uint8_t *state_mem = new uint8_t[state_size];
-
-  {
-    FILE *fp_read = fopen(statefile, modes);
-    if (state_size != llama_get_state_size(constState)) {
-      fprintf(stderr, "\n%s : failed to validate state size\n", __func__);
-      return 1;
-    }
-
-    const size_t ret = fread(state_mem, 1, state_size, fp_read);
-    if (ret != state_size) {
-      fprintf(stderr, "\n%s : failed to read state\n", __func__);
-      return 1;
-    }
-
-    llama_set_state_data(
-        state, state_mem); // could also read directly from memory mapped file
-    fclose(fp_read);
-  }
-
-  return 0;
-}
-
-void save_state(void *ctx, char *dst, char *modes) {
-  llama_context *state = (llama_context *)ctx;
-
-  const size_t state_size = llama_get_state_size(state);
-  uint8_t *state_mem = new uint8_t[state_size];
-
-  // Save state (rng, logits, embedding and kv_cache) to file
-  {
-    FILE *fp_write = fopen(dst, modes);
-    llama_copy_state_data(
-        state, state_mem); // could also copy directly to memory mapped file
-    fwrite(state_mem, 1, state_size, fp_write);
-    fclose(fp_write);
-  }
-}
-
-void *llama_allocate_params(
-    const char *prompt, int seed, int threads, int tokens, int top_k,
-    float top_p, float temp, float repeat_penalty, int repeat_last_n,
-    bool ignore_eos, bool memory_f16, int n_batch, int n_keep,
-    const char **antiprompt, int antiprompt_count, float tfs_z, float typical_p,
-    float frequency_penalty, float presence_penalty, int mirostat,
-    float mirostat_eta, float mirostat_tau, bool penalize_nl,
-    const char *logit_bias, bool mlock, bool mmap, const char *maingpu,
-    const char *tensorsplit) {
-  gpt_params *params = new gpt_params;
-  params->seed = seed;
-  params->n_threads = threads;
-  params->n_predict = tokens;
-  params->repeat_last_n = repeat_last_n;
-  params->top_k = top_k;
-  params->top_p = top_p;
-  params->memory_f16 = memory_f16;
-  params->temp = temp;
-  params->use_mmap = mmap;
-  params->use_mlock = mlock;
-  params->repeat_penalty = repeat_penalty;
-  params->n_batch = n_batch;
-  params->n_keep = n_keep;
-  if (maingpu[0] != '\0') {
-    params->main_gpu = std::stoi(maingpu);
-  }
-
-  if (tensorsplit[0] != '\0') {
-    std::string arg_next = tensorsplit;
-    // split string by , and /
-    const std::regex regex{R"([,/]+)"};
-    std::sregex_token_iterator it{arg_next.begin(), arg_next.end(), regex, -1};
-    std::vector<std::string> split_arg{it, {}};
-    GGML_ASSERT(split_arg.size() <= LLAMA_MAX_DEVICES);
-
-    for (size_t i = 0; i < LLAMA_MAX_DEVICES; ++i) {
-      if (i < split_arg.size()) {
-        params->tensor_split[i] = std::stof(split_arg[i]);
-      } else {
-        params->tensor_split[i] = 0.0f;
-      }
-    }
-  }
-
-  if (ignore_eos) {
-    params->logit_bias[llama_token_eos()] = -INFINITY;
-  }
-
-  for (int i = 0; i < antiprompt_count; i++) {
-    params->antiprompt.push_back(antiprompt[i]);
-  }
-
-  params->tfs_z = tfs_z;
-  params->typical_p = typical_p;
-  params->presence_penalty = presence_penalty;
-  params->mirostat = mirostat;
-  params->mirostat_eta = mirostat_eta;
-  params->mirostat_tau = mirostat_tau;
-  params->penalize_nl = penalize_nl;
-  std::stringstream ss(logit_bias);
-  llama_token key;
-  char sign;
-  std::string value_str;
-  if (ss >> key && ss >> sign && std::getline(ss, value_str) &&
-      (sign == '+' || sign == '-')) {
-    params->logit_bias[key] =
-        std::stof(value_str) * ((sign == '-') ? -1.0f : 1.0f);
-  }
-  params->frequency_penalty = frequency_penalty;
-  params->prompt = prompt;
-
-  return params;
-}
-
-void *load_model(const char *fname, int n_ctx, int n_seed, bool memory_f16,
-                 bool mlock, bool embeddings, bool mmap, bool low_vram,
-                 bool vocab_only, int n_gpu_layers, int n_batch,
-                 const char *maingpu, const char *tensorsplit, bool numa) {
-  // load the model
-  auto lparams = llama_context_default_params();
-
-  lparams.n_ctx = n_ctx;
-  lparams.seed = n_seed;
-  lparams.f16_kv = memory_f16;
-  lparams.embedding = embeddings;
-  lparams.use_mlock = mlock;
-  lparams.n_gpu_layers = n_gpu_layers;
-  lparams.use_mmap = mmap;
-  lparams.low_vram = low_vram;
-  lparams.vocab_only = vocab_only;
-
-  if (maingpu[0] != '\0') {
-    lparams.main_gpu = std::stoi(maingpu);
-  }
-
-  if (tensorsplit[0] != '\0') {
-    std::string arg_next = tensorsplit;
-    // split string by , and /
-    const std::regex regex{R"([,/]+)"};
-    std::sregex_token_iterator it{arg_next.begin(), arg_next.end(), regex, -1};
-    std::vector<std::string> split_arg{it, {}};
-    GGML_ASSERT(split_arg.size() <= LLAMA_MAX_DEVICES);
-
-    for (size_t i = 0; i < LLAMA_MAX_DEVICES; ++i) {
-      if (i < split_arg.size()) {
-        lparams.tensor_split[i] = std::stof(split_arg[i]);
-      } else {
-        lparams.tensor_split[i] = 0.0f;
-      }
-    }
-  }
-
-  lparams.n_batch = n_batch;
-
-  llama_init_backend(numa);
-  void *res = nullptr;
-  try {
-    res = llama_init_from_file(fname, lparams);
-  } catch (std::runtime_error &e) {
-    fprintf(stderr, "failed %s", e.what());
-    return res;
-  }
-
-  return res;
-}
--- a/llama/binding/binding.h
+++ b/llama/binding/binding.h
@@ -1,69 +0,0 @@
-// MIT License
-
-// Copyright (c) 2023 go-skynet authors
-
-// Permission is hereby granted, free of charge, to any person obtaining a copy
-// of this software and associated documentation files (the "Software"), to deal
-// in the Software without restriction, including without limitation the rights
-// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
-// copies of the Software, and to permit persons to whom the Software is
-// furnished to do so, subject to the following conditions:
-
-// The above copyright notice and this permission notice shall be included in all
-// copies or substantial portions of the Software.
-
-// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
-// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
-// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
-// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
-// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
-// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
-// SOFTWARE.
-
-#ifdef __cplusplus
-
-extern "C" {
-
-#endif
-
-#include <stdbool.h>
-
-extern unsigned char tokenCallback(void *, char *);
-
-int load_state(void *ctx, char *statefile, char *modes);
-
-int eval(void *params_ptr, void *ctx, char *text);
-
-void save_state(void *ctx, char *dst, char *modes);
-
-void *load_model(const char *fname, int n_ctx, int n_seed, bool memory_f16,
-                 bool mlock, bool embeddings, bool mmap, bool low_vram,
-                 bool vocab_only, int n_gpu, int n_batch, const char *maingpu,
-                 const char *tensorsplit, bool numa);
-
-int get_embeddings(void *params_ptr, void *state_pr, float *res_embeddings);
-
-int get_token_embeddings(void *params_ptr, void *state_pr, int *tokens,
-                         int tokenSize, float *res_embeddings);
-
-void *llama_allocate_params(
-    const char *prompt, int seed, int threads, int tokens, int top_k,
-    float top_p, float temp, float repeat_penalty, int repeat_last_n,
-    bool ignore_eos, bool memory_f16, int n_batch, int n_keep,
-    const char **antiprompt, int antiprompt_count, float tfs_z, float typical_p,
-    float frequency_penalty, float presence_penalty, int mirostat,
-    float mirostat_eta, float mirostat_tau, bool penalize_nl,
-    const char *logit_bias, bool mlock, bool mmap, const char *maingpu,
-    const char *tensorsplit);
-
-void llama_free_params(void *params_ptr);
-
-void llama_binding_free_model(void *state);
-
-int llama_predict(void *params_ptr, void *state_pr, char *result, bool debug);
-
-#ifdef __cplusplus
-
-}
-
-#endif
--- a/llama/llama.go
+++ b/llama/llama.go
@@ -1,215 +0,0 @@
-// MIT License
-
-// Copyright (c) 2023 go-skynet authors
-
-// Permission is hereby granted, free of charge, to any person obtaining a copy
-// of this software and associated documentation files (the "Software"), to deal
-// in the Software without restriction, including without limitation the rights
-// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
-// copies of the Software, and to permit persons to whom the Software is
-// furnished to do so, subject to the following conditions:
-
-// The above copyright notice and this permission notice shall be included in all
-// copies or substantial portions of the Software.
-
-// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
-// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
-// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
-// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
-// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
-// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
-// SOFTWARE.
-
-package llama
-
-// #cgo LDFLAGS: -Lbuild -lbinding -lllama -lm -lggml_static -lstdc++
-// #cgo CXXFLAGS: -std=c++11
-// #cgo darwin LDFLAGS: -framework Accelerate -framework Foundation -framework Metal -framework MetalKit -framework MetalPerformanceShaders
-// #include "binding/binding.h"
-// #include <stdlib.h>
-import "C"
-
-import (
-	"fmt"
-	"strings"
-	"sync"
-	"unsafe"
-)
-
-type LLama struct {
-	ctx         unsafe.Pointer
-	embeddings  bool
-	contextSize int
-}
-
-func New(model string, mo ModelOptions) (*LLama, error) {
-	modelPath := C.CString(model)
-	defer C.free(unsafe.Pointer(modelPath))
-
-	ctx := C.load_model(modelPath, C.int(mo.ContextSize), C.int(mo.Seed), C.bool(mo.F16Memory), C.bool(mo.MLock), C.bool(mo.Embeddings), C.bool(mo.MMap), C.bool(mo.LowVRAM), C.bool(mo.VocabOnly), C.int(mo.NGPULayers), C.int(mo.NBatch), C.CString(mo.MainGPU), C.CString(mo.TensorSplit), C.bool(mo.NUMA))
-	if ctx == nil {
-		return nil, fmt.Errorf("failed loading model")
-	}
-
-	ll := &LLama{ctx: ctx, contextSize: mo.ContextSize, embeddings: mo.Embeddings}
-
-	return ll, nil
-}
-
-func (l *LLama) Free() {
-	C.llama_binding_free_model(l.ctx)
-}
-
-func (l *LLama) Eval(text string, po PredictOptions) error {
-	input := C.CString(text)
-	if po.Tokens == 0 {
-		po.Tokens = 99999999
-	}
-	defer C.free(unsafe.Pointer(input))
-
-	reverseCount := len(po.StopPrompts)
-	reversePrompt := make([]*C.char, reverseCount)
-	var pass **C.char
-	for i, s := range po.StopPrompts {
-		cs := C.CString(s)
-		reversePrompt[i] = cs
-		pass = &reversePrompt[0]
-		defer C.free(unsafe.Pointer(cs))
-	}
-
-	cLogitBias := C.CString(po.LogitBias)
-	defer C.free(unsafe.Pointer(cLogitBias))
-
-	cMainGPU := C.CString(po.MainGPU)
-	defer C.free(unsafe.Pointer(cMainGPU))
-
-	cTensorSplit := C.CString(po.TensorSplit)
-	defer C.free(unsafe.Pointer(cTensorSplit))
-
-	params := C.llama_allocate_params(input, C.int(po.Seed), C.int(po.Threads), C.int(po.Tokens), C.int(po.TopK),
-		C.float(po.TopP), C.float(po.Temperature), C.float(po.Penalty), C.int(po.Repeat),
-		C.bool(po.IgnoreEOS), C.bool(po.F16KV),
-		C.int(po.Batch), C.int(po.NKeep), pass, C.int(reverseCount),
-		C.float(po.TailFreeSamplingZ), C.float(po.TypicalP), C.float(po.FrequencyPenalty), C.float(po.PresencePenalty),
-		C.int(po.Mirostat), C.float(po.MirostatETA), C.float(po.MirostatTAU), C.bool(po.PenalizeNL), cLogitBias,
-		C.bool(po.MLock), C.bool(po.MMap), cMainGPU, cTensorSplit,
-	)
-	defer C.llama_free_params(params)
-
-	ret := C.eval(params, l.ctx, input)
-	if ret != 0 {
-		return fmt.Errorf("inference failed")
-	}
-
-	return nil
-}
-
-func (l *LLama) Predict(text string, po PredictOptions) (string, error) {
-	if po.TokenCallback != nil {
-		setCallback(l.ctx, po.TokenCallback)
-	}
-
-	input := C.CString(text)
-	if po.Tokens == 0 {
-		po.Tokens = 99999999
-	}
-	defer C.free(unsafe.Pointer(input))
-
-	out := make([]byte, po.Tokens)
-
-	reverseCount := len(po.StopPrompts)
-	reversePrompt := make([]*C.char, reverseCount)
-	var pass **C.char
-	for i, s := range po.StopPrompts {
-		cs := C.CString(s)
-		reversePrompt[i] = cs
-		pass = &reversePrompt[0]
-		defer C.free(unsafe.Pointer(cs))
-	}
-
-	cLogitBias := C.CString(po.LogitBias)
-	defer C.free(unsafe.Pointer(cLogitBias))
-
-	cMainGPU := C.CString(po.MainGPU)
-	defer C.free(unsafe.Pointer(cMainGPU))
-
-	cTensorSplit := C.CString(po.TensorSplit)
-	defer C.free(unsafe.Pointer(cTensorSplit))
-
-	params := C.llama_allocate_params(input, C.int(po.Seed), C.int(po.Threads), C.int(po.Tokens), C.int(po.TopK),
-		C.float(po.TopP), C.float(po.Temperature), C.float(po.Penalty), C.int(po.Repeat),
-		C.bool(po.IgnoreEOS), C.bool(po.F16KV),
-		C.int(po.Batch), C.int(po.NKeep), pass, C.int(reverseCount),
-		C.float(po.TailFreeSamplingZ), C.float(po.TypicalP), C.float(po.FrequencyPenalty), C.float(po.PresencePenalty),
-		C.int(po.Mirostat), C.float(po.MirostatETA), C.float(po.MirostatTAU), C.bool(po.PenalizeNL), cLogitBias,
-		C.bool(po.MLock), C.bool(po.MMap), cMainGPU, cTensorSplit,
-	)
-	defer C.llama_free_params(params)
-
-	ret := C.llama_predict(params, l.ctx, (*C.char)(unsafe.Pointer(&out[0])), C.bool(po.DebugMode))
-	if ret != 0 {
-		return "", fmt.Errorf("inference failed")
-	}
-	res := C.GoString((*C.char)(unsafe.Pointer(&out[0])))
-
-	res = strings.TrimPrefix(res, " ")
-	res = strings.TrimPrefix(res, text)
-	res = strings.TrimPrefix(res, "\n")
-
-	for _, s := range po.StopPrompts {
-		res = strings.TrimRight(res, s)
-	}
-
-	if po.TokenCallback != nil {
-		setCallback(l.ctx, nil)
-	}
-
-	return res, nil
-}
-
-// CGo only allows us to use static calls from C to Go, we can't just dynamically pass in func's.
-// This is the next best thing, we register the callbacks in this map and call tokenCallback from
-// the C code. We also attach a finalizer to LLama, so it will unregister the callback when the
-// garbage collection frees it.
-
-// SetTokenCallback registers a callback for the individual tokens created when running Predict. It
-// will be called once for each token. The callback shall return true as long as the model should
-// continue predicting the next token. When the callback returns false the predictor will return.
-// The tokens are just converted into Go strings, they are not trimmed or otherwise changed. Also
-// the tokens may not be valid UTF-8.
-// Pass in nil to remove a callback.
-//
-// It is save to call this method while a prediction is running.
-func (l *LLama) SetTokenCallback(callback func(token string) bool) {
-	setCallback(l.ctx, callback)
-}
-
-var (
-	m         sync.Mutex
-	callbacks = map[uintptr]func(string) bool{}
-)
-
-//export tokenCallback
-func tokenCallback(statePtr unsafe.Pointer, token *C.char) bool {
-	m.Lock()
-	defer m.Unlock()
-
-	if callback, ok := callbacks[uintptr(statePtr)]; ok {
-		return callback(C.GoString(token))
-	}
-
-	return true
-}
-
-// setCallback can be used to register a token callback for LLama. Pass in a nil callback to
-// remove the callback.
-func setCallback(statePtr unsafe.Pointer, callback func(string) bool) {
-	m.Lock()
-	defer m.Unlock()
-
-	if callback == nil {
-		delete(callbacks, uintptr(statePtr))
-	} else {
-		callbacks[uintptr(statePtr)] = callback
-	}
-}
--- a/llama/llama_cublas.go
+++ b/llama/llama_cublas.go
@@ -1,9 +0,0 @@
-//go:build cublas
-// +build cublas
-
-package llama
-
-/*
-#cgo LDFLAGS: -lcublas -lcudart -L/usr/local/cuda/lib64/
-*/
-import "C"
--- a/llama/llama_metal.go
+++ b/llama/llama_metal.go
@@ -1,2 +0,0 @@
-//go:build metal
-package llama
--- a/llama/llama_openblas.go
+++ b/llama/llama_openblas.go
@@ -1,9 +0,0 @@
-//go:build openblas
-// +build openblas
-
-package llama
-
-/*
-#cgo LDFLAGS: -lopenblas
-*/
-import "C"
--- a/llama/options.go
+++ b/llama/options.go
@@ -1,98 +0,0 @@
-// MIT License
-
-// Copyright (c) 2023 go-skynet authors
-
-// Permission is hereby granted, free of charge, to any person obtaining a copy
-// of this software and associated documentation files (the "Software"), to deal
-// in the Software without restriction, including without limitation the rights
-// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
-// copies of the Software, and to permit persons to whom the Software is
-// furnished to do so, subject to the following conditions:
-
-// The above copyright notice and this permission notice shall be included in all
-// copies or substantial portions of the Software.
-
-// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
-// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
-// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
-// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
-// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
-// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
-// SOFTWARE.
-
-package llama
-
-type ModelOptions struct {
-	ContextSize int
-	Seed        int
-	NBatch      int
-	F16Memory   bool
-	MLock       bool
-	MMap        bool
-	VocabOnly   bool
-	LowVRAM     bool
-	Embeddings  bool
-	NUMA        bool
-	NGPULayers  int
-	MainGPU     string
-	TensorSplit string
-}
-
-type PredictOptions struct {
-	Seed, Threads, Tokens, TopK, Repeat, Batch, NKeep int
-	TopP, Temperature, Penalty                        float64
-	F16KV                                             bool
-	DebugMode                                         bool
-	StopPrompts                                       []string
-	IgnoreEOS                                         bool
-
-	TailFreeSamplingZ float64
-	TypicalP          float64
-	FrequencyPenalty  float64
-	PresencePenalty   float64
-	Mirostat          int
-	MirostatETA       float64
-	MirostatTAU       float64
-	PenalizeNL        bool
-	LogitBias         string
-	TokenCallback     func(string) bool
-
-	MLock, MMap bool
-	MainGPU     string
-	TensorSplit string
-}
-
-type PredictOption func(p *PredictOptions)
-
-type ModelOption func(p *ModelOptions)
-
-var DefaultModelOptions ModelOptions = ModelOptions{
-	ContextSize: 512,
-	Seed:        0,
-	F16Memory:   false,
-	MLock:       false,
-	Embeddings:  false,
-	MMap:        true,
-	LowVRAM:     false,
-}
-
-var DefaultOptions PredictOptions = PredictOptions{
-	Seed:              -1,
-	Threads:           4,
-	Tokens:            128,
-	Penalty:           1.1,
-	Repeat:            64,
-	Batch:             512,
-	NKeep:             64,
-	TopK:              40,
-	TopP:              0.95,
-	TailFreeSamplingZ: 1.0,
-	TypicalP:          1.0,
-	Temperature:       0.8,
-	FrequencyPenalty:  0.0,
-	PresencePenalty:   0.0,
-	Mirostat:          0,
-	MirostatTAU:       5.0,
-	MirostatETA:       0.1,
-	MMap:              true,
-}
--- a/llm/falcon.go
+++ b/llm/falcon.go
@@ -0,0 +1,22 @@
+package llm
+
+const ModelFamilyFalcon = "falcon"
+
+const (
+	falconModelType7B   = 32
+	falconModelType40B  = 60
+	falconModelType180B = 80
+)
+
+func falconModelType(numLayer uint32) string {
+	switch numLayer {
+	case 32:
+		return "7B"
+	case 60:
+		return "40B"
+	case 80:
+		return "180B"
+	default:
+		return "Unknown"
+	}
+}
--- a/llm/ggml.go
+++ b/llm/ggml.go
@@ -0,0 +1,209 @@
+package llm
+
+import (
+	"encoding/binary"
+	"errors"
+	"io"
+)
+
+type GGML struct {
+	magic uint32
+	container
+	model
+}
+
+const (
+	fileTypeF32 uint32 = iota
+	fileTypeF16
+	fileTypeQ4_0
+	fileTypeQ4_1
+	fileTypeQ4_1_F16
+	fileTypeQ8_0 uint32 = iota + 2
+	fileTypeQ5_0
+	fileTypeQ5_1
+	fileTypeQ2_K
+	fileTypeQ3_K_S
+	fileTypeQ3_K_M
+	fileTypeQ3_K_L
+	fileTypeQ4_K_S
+	fileTypeQ4_K_M
+	fileTypeQ5_K_S
+	fileTypeQ5_K_M
+	fileTypeQ6_K
+)
+
+func fileType(fileType uint32) string {
+	switch fileType {
+	case fileTypeF32:
+		return "F32"
+	case fileTypeF16:
+		return "F16"
+	case fileTypeQ4_0:
+		return "Q4_0"
+	case fileTypeQ4_1:
+		return "Q4_1"
+	case fileTypeQ4_1_F16:
+		return "Q4_1_F16"
+	case fileTypeQ8_0:
+		return "Q8_0"
+	case fileTypeQ5_0:
+		return "Q5_0"
+	case fileTypeQ5_1:
+		return "Q5_1"
+	case fileTypeQ2_K:
+		return "Q2_K"
+	case fileTypeQ3_K_S:
+		return "Q3_K_S"
+	case fileTypeQ3_K_M:
+		return "Q3_K_M"
+	case fileTypeQ3_K_L:
+		return "Q3_K_L"
+	case fileTypeQ4_K_S:
+		return "Q4_K_S"
+	case fileTypeQ4_K_M:
+		return "Q4_K_M"
+	case fileTypeQ5_K_S:
+		return "Q5_K_S"
+	case fileTypeQ5_K_M:
+		return "Q5_K_M"
+	case fileTypeQ6_K:
+		return "Q6_K"
+	default:
+		return "Unknown"
+	}
+}
+
+type model interface {
+	ModelFamily() string
+	ModelType() string
+	FileType() string
+	NumLayers() int64
+}
+
+type container interface {
+	Name() string
+	Decode(io.Reader) (model, error)
+}
+
+type containerGGML struct{}
+
+func (c *containerGGML) Name() string {
+	return "ggml"
+}
+
+func (c *containerGGML) Decode(r io.Reader) (model, error) {
+	return nil, nil
+}
+
+type containerGGMF struct {
+	version uint32
+}
+
+func (c *containerGGMF) Name() string {
+	return "ggmf"
+}
+
+func (c *containerGGMF) Decode(r io.Reader) (model, error) {
+	var version uint32
+	binary.Read(r, binary.LittleEndian, &version)
+
+	switch version {
+	case 1:
+	default:
+		return nil, errors.New("invalid version")
+	}
+
+	c.version = version
+	return nil, nil
+}
+
+type containerGGJT struct {
+	version uint32
+}
+
+func (c *containerGGJT) Name() string {
+	return "ggjt"
+}
+
+func (c *containerGGJT) Decode(r io.Reader) (model, error) {
+	var version uint32
+	binary.Read(r, binary.LittleEndian, &version)
+
+	switch version {
+	case 1, 2, 3:
+	default:
+		return nil, errors.New("invalid version")
+	}
+
+	c.version = version
+
+	// different model types may have different layouts for hyperparameters
+	var llama llamaModel
+	binary.Read(r, binary.LittleEndian, &llama.hyperparameters)
+	return &llama, nil
+}
+
+type containerLORA struct {
+	version uint32
+}
+
+func (c *containerLORA) Name() string {
+	return "ggla"
+}
+
+func (c *containerLORA) Decode(r io.Reader) (model, error) {
+	var version uint32
+	binary.Read(r, binary.LittleEndian, &version)
+
+	switch version {
+	case 1:
+	default:
+		return nil, errors.New("invalid version")
+	}
+
+	c.version = version
+	return nil, nil
+}
+
+const (
+	// Magic constant for `ggml` files (unversioned).
+	FILE_MAGIC_GGML = 0x67676d6c
+	// Magic constant for `ggml` files (versioned, ggmf).
+	FILE_MAGIC_GGMF = 0x67676d66
+	// Magic constant for `ggml` files (versioned, ggjt).
+	FILE_MAGIC_GGJT = 0x67676a74
+	// Magic constant for `ggla` files (LoRA adapter).
+	FILE_MAGIC_GGLA = 0x67676C61
+	// Magic constant for `gguf` files (versioned, gguf)
+	FILE_MAGIC_GGUF = 0x46554747
+)
+
+func DecodeGGML(r io.ReadSeeker) (*GGML, error) {
+	var ggml GGML
+	binary.Read(r, binary.LittleEndian, &ggml.magic)
+
+	switch ggml.magic {
+	case FILE_MAGIC_GGML:
+		ggml.container = &containerGGML{}
+	case FILE_MAGIC_GGMF:
+		ggml.container = &containerGGMF{}
+	case FILE_MAGIC_GGJT:
+		ggml.container = &containerGGJT{}
+	case FILE_MAGIC_GGLA:
+		ggml.container = &containerLORA{}
+	case FILE_MAGIC_GGUF:
+		ggml.container = &containerGGUF{}
+	default:
+		return nil, errors.New("invalid file magic")
+	}
+
+	model, err := ggml.Decode(r)
+	if err != nil {
+		return nil, err
+	}
+
+	ggml.model = model
+
+	// final model type
+	return &ggml, nil
+}
--- a/llm/gguf.go
+++ b/llm/gguf.go
@@ -0,0 +1,379 @@
+package llm
+
+import (
+	"bytes"
+	"encoding/binary"
+	"errors"
+	"fmt"
+	"io"
+)
+
+type containerGGUF struct {
+	Version uint32
+
+	V1 struct {
+		NumTensor uint32
+		NumKV     uint32
+	}
+
+	V2 struct {
+		NumTensor uint64
+		NumKV     uint64
+	}
+}
+
+func (c *containerGGUF) Name() string {
+	return "gguf"
+}
+
+func (c *containerGGUF) Decode(r io.Reader) (model, error) {
+	binary.Read(r, binary.LittleEndian, &c.Version)
+
+	switch c.Version {
+	case 1:
+		binary.Read(r, binary.LittleEndian, &c.V1)
+	case 2:
+		binary.Read(r, binary.LittleEndian, &c.V2)
+	default:
+		return nil, errors.New("invalid version")
+	}
+
+	model := newGGUFModel(c)
+	if err := model.Decode(r); err != nil {
+		return nil, err
+	}
+
+	return model, nil
+}
+
+const (
+	ggufTypeUint8 uint32 = iota
+	ggufTypeInt8
+	ggufTypeUint16
+	ggufTypeInt16
+	ggufTypeUint32
+	ggufTypeInt32
+	ggufTypeFloat32
+	ggufTypeBool
+	ggufTypeString
+	ggufTypeArray
+	ggufTypeUint64
+	ggufTypeInt64
+	ggufTypeFloat64
+)
+
+type kv map[string]any
+
+type ggufModel struct {
+	*containerGGUF
+	kv
+}
+
+func newGGUFModel(container *containerGGUF) *ggufModel {
+	return &ggufModel{
+		containerGGUF: container,
+		kv:            make(kv),
+	}
+}
+
+func (llm *ggufModel) NumKV() uint64 {
+	if llm.Version == 1 {
+		return uint64(llm.V1.NumKV)
+	}
+
+	return llm.V2.NumKV
+}
+
+func (llm *ggufModel) ModelFamily() string {
+	t, ok := llm.kv["general.architecture"].(string)
+	if ok {
+		return t
+	}
+
+	return "unknown"
+}
+
+func (llm *ggufModel) ModelType() string {
+	switch llm.ModelFamily() {
+	case "llama":
+		if blocks, ok := llm.kv["llama.block_count"].(uint32); ok {
+			heads, headsOK := llm.kv["llama.head_count"].(uint32)
+			headKVs, headsKVsOK := llm.kv["llama.head_count_kv"].(uint32)
+			if headsOK && headsKVsOK && heads/headKVs == 8 {
+				return "70B"
+			}
+
+			return llamaModelType(blocks)
+		}
+	case "falcon":
+		if blocks, ok := llm.kv["falcon.block_count"].(uint32); ok {
+			return falconModelType(blocks)
+		}
+	}
+
+	return "Unknown"
+}
+
+func (llm *ggufModel) FileType() string {
+	t, ok := llm.kv["general.file_type"].(uint32)
+	if ok {
+		return fileType(t)
+	}
+
+	return "Unknown"
+}
+
+func (llm *ggufModel) Decode(r io.Reader) error {
+	read := llm.readString
+	if llm.Version == 1 {
+		read = llm.readStringV1
+	}
+
+	for i := 0; uint64(i) < llm.NumKV(); i++ {
+		k, err := read(r)
+		if err != nil {
+			return err
+		}
+
+		vtype := llm.readU32(r)
+
+		var v any
+		switch vtype {
+		case ggufTypeUint8:
+			v = llm.readU8(r)
+		case ggufTypeInt8:
+			v = llm.readI8(r)
+		case ggufTypeUint16:
+			v = llm.readU16(r)
+		case ggufTypeInt16:
+			v = llm.readI16(r)
+		case ggufTypeUint32:
+			v = llm.readU32(r)
+		case ggufTypeInt32:
+			v = llm.readI32(r)
+		case ggufTypeUint64:
+			v = llm.readU64(r)
+		case ggufTypeInt64:
+			v = llm.readI64(r)
+		case ggufTypeFloat32:
+			v = llm.readF32(r)
+		case ggufTypeFloat64:
+			v = llm.readF64(r)
+		case ggufTypeBool:
+			v = llm.readBool(r)
+		case ggufTypeString:
+			fn := llm.readString
+			if llm.Version == 1 {
+				fn = llm.readStringV1
+			}
+
+			s, err := fn(r)
+			if err != nil {
+				return err
+			}
+
+			v = s
+		case ggufTypeArray:
+			fn := llm.readArray
+			if llm.Version == 1 {
+				fn = llm.readArrayV1
+			}
+
+			a, err := fn(r)
+			if err != nil {
+				return err
+			}
+
+			v = a
+		default:
+			return fmt.Errorf("invalid type: %d", vtype)
+		}
+
+		llm.kv[k] = v
+	}
+
+	return nil
+}
+
+func (llm *ggufModel) NumLayers() int64 {
+	value, exists := llm.kv[fmt.Sprintf("%s.block_count", llm.ModelFamily())]
+	if !exists {
+		return 0
+	}
+
+	v := value.(uint32)
+	return int64(v)
+}
+
+func (ggufModel) readU8(r io.Reader) uint8 {
+	var u8 uint8
+	binary.Read(r, binary.LittleEndian, &u8)
+	return u8
+}
+
+func (ggufModel) readI8(r io.Reader) int8 {
+	var i8 int8
+	binary.Read(r, binary.LittleEndian, &i8)
+	return i8
+}
+
+func (ggufModel) readU16(r io.Reader) uint16 {
+	var u16 uint16
+	binary.Read(r, binary.LittleEndian, &u16)
+	return u16
+}
+
+func (ggufModel) readI16(r io.Reader) int16 {
+	var i16 int16
+	binary.Read(r, binary.LittleEndian, &i16)
+	return i16
+}
+
+func (ggufModel) readU32(r io.Reader) uint32 {
+	var u32 uint32
+	binary.Read(r, binary.LittleEndian, &u32)
+	return u32
+}
+
+func (ggufModel) readI32(r io.Reader) int32 {
+	var i32 int32
+	binary.Read(r, binary.LittleEndian, &i32)
+	return i32
+}
+
+func (ggufModel) readU64(r io.Reader) uint64 {
+	var u64 uint64
+	binary.Read(r, binary.LittleEndian, &u64)
+	return u64
+}
+
+func (ggufModel) readI64(r io.Reader) int64 {
+	var i64 int64
+	binary.Read(r, binary.LittleEndian, &i64)
+	return i64
+}
+
+func (ggufModel) readF32(r io.Reader) float32 {
+	var f32 float32
+	binary.Read(r, binary.LittleEndian, &f32)
+	return f32
+}
+
+func (ggufModel) readF64(r io.Reader) float64 {
+	var f64 float64
+	binary.Read(r, binary.LittleEndian, &f64)
+	return f64
+}
+
+func (ggufModel) readBool(r io.Reader) bool {
+	var b bool
+	binary.Read(r, binary.LittleEndian, &b)
+	return b
+}
+
+func (ggufModel) readStringV1(r io.Reader) (string, error) {
+	var nameLength uint32
+	binary.Read(r, binary.LittleEndian, &nameLength)
+
+	var b bytes.Buffer
+	if _, err := io.CopyN(&b, r, int64(nameLength)); err != nil {
+		return "", err
+	}
+
+	// gguf v1 strings are null-terminated
+	b.Truncate(b.Len() - 1)
+
+	return b.String(), nil
+}
+
+func (llm ggufModel) readString(r io.Reader) (string, error) {
+	var nameLength uint64
+	binary.Read(r, binary.LittleEndian, &nameLength)
+
+	var b bytes.Buffer
+	if _, err := io.CopyN(&b, r, int64(nameLength)); err != nil {
+		return "", err
+	}
+
+	return b.String(), nil
+}
+
+func (llm *ggufModel) readArrayV1(r io.Reader) (arr []any, err error) {
+	atype := llm.readU32(r)
+	n := llm.readU32(r)
+
+	for i := 0; uint32(i) < n; i++ {
+		switch atype {
+		case ggufTypeUint8:
+			arr = append(arr, llm.readU8(r))
+		case ggufTypeInt8:
+			arr = append(arr, llm.readU8(r))
+		case ggufTypeUint16:
+			arr = append(arr, llm.readU16(r))
+		case ggufTypeInt16:
+			arr = append(arr, llm.readI16(r))
+		case ggufTypeUint32:
+			arr = append(arr, llm.readU32(r))
+		case ggufTypeInt32:
+			arr = append(arr, llm.readI32(r))
+		case ggufTypeFloat32:
+			arr = append(arr, llm.readF32(r))
+		case ggufTypeBool:
+			arr = append(arr, llm.readBool(r))
+		case ggufTypeString:
+			s, err := llm.readStringV1(r)
+			if err != nil {
+				return nil, err
+			}
+
+			arr = append(arr, s)
+		default:
+			return nil, fmt.Errorf("invalid array type: %d", atype)
+		}
+	}
+
+	return
+}
+
+func (llm *ggufModel) readArray(r io.Reader) (arr []any, err error) {
+	atype := llm.readU32(r)
+	n := llm.readU64(r)
+
+	for i := 0; uint64(i) < n; i++ {
+		switch atype {
+		case ggufTypeUint8:
+			arr = append(arr, llm.readU8(r))
+		case ggufTypeInt8:
+			arr = append(arr, llm.readU8(r))
+		case ggufTypeUint16:
+			arr = append(arr, llm.readU16(r))
+		case ggufTypeInt16:
+			arr = append(arr, llm.readI16(r))
+		case ggufTypeUint32:
+			arr = append(arr, llm.readU32(r))
+		case ggufTypeInt32:
+			arr = append(arr, llm.readI32(r))
+		case ggufTypeUint64:
+			arr = append(arr, llm.readU64(r))
+		case ggufTypeInt64:
+			arr = append(arr, llm.readI64(r))
+		case ggufTypeFloat32:
+			arr = append(arr, llm.readF32(r))
+		case ggufTypeFloat64:
+			arr = append(arr, llm.readF64(r))
+		case ggufTypeBool:
+			arr = append(arr, llm.readBool(r))
+		case ggufTypeString:
+			s, err := llm.readString(r)
+			if err != nil {
+				return nil, err
+			}
+
+			arr = append(arr, s)
+		default:
+			return nil, fmt.Errorf("invalid array type: %d", atype)
+		}
+	}
+
+	return
+}
--- a/llm/llama.cpp/generate_darwin_amd64.go
+++ b/llm/llama.cpp/generate_darwin_amd64.go
@@ -0,0 +1,16 @@
+package llm
+
+//go:generate git submodule init
+
+//go:generate git submodule update --force ggml
+//go:generate git -C ggml apply ../patches/0001-add-detokenize-endpoint.patch
+//go:generate git -C ggml apply ../patches/0002-34B-model-support.patch
+//go:generate git -C ggml apply ../patches/0003-metal-fix-synchronization-in-new-matrix-multiplicati.patch
+//go:generate git -C ggml apply ../patches/0004-metal-add-missing-barriers-for-mul-mat-2699.patch
+//go:generate cmake -S ggml -B ggml/build/cpu -DLLAMA_ACCELERATE=on -DLLAMA_K_QUANTS=on -DCMAKE_SYSTEM_PROCESSOR=x86_64 -DCMAKE_OSX_ARCHITECTURES=x86_64 -DCMAKE_OSX_DEPLOYMENT_TARGET=11.0
+//go:generate cmake --build ggml/build/cpu --target server --config Release
+
+//go:generate git submodule update --force gguf
+//go:generate git -C gguf apply ../patches/0001-remove-warm-up-logging.patch
+//go:generate cmake -S gguf -B gguf/build/cpu -DLLAMA_ACCELERATE=on -DLLAMA_K_QUANTS=on -DCMAKE_SYSTEM_PROCESSOR=x86_64 -DCMAKE_OSX_ARCHITECTURES=x86_64 -DCMAKE_OSX_DEPLOYMENT_TARGET=11.0
+//go:generate cmake --build gguf/build/cpu --target server --config Release
--- a/llm/llama.cpp/generate_darwin_arm64.go
+++ b/llm/llama.cpp/generate_darwin_arm64.go
@@ -0,0 +1,16 @@
+package llm
+
+//go:generate git submodule init
+
+//go:generate git submodule update --force ggml
+//go:generate git -C ggml apply ../patches/0001-add-detokenize-endpoint.patch
+//go:generate git -C ggml apply ../patches/0002-34B-model-support.patch
+//go:generate git -C ggml apply ../patches/0003-metal-fix-synchronization-in-new-matrix-multiplicati.patch
+//go:generate git -C ggml apply ../patches/0004-metal-add-missing-barriers-for-mul-mat-2699.patch
+//go:generate cmake -S ggml -B ggml/build/metal -DLLAMA_METAL=on -DLLAMA_ACCELERATE=on -DLLAMA_K_QUANTS=on -DCMAKE_SYSTEM_PROCESSOR=arm64 -DCMAKE_OSX_ARCHITECTURES=arm64 -DCMAKE_OSX_DEPLOYMENT_TARGET=11.0
+//go:generate cmake --build ggml/build/metal --target server --config Release
+
+//go:generate git submodule update --force gguf
+//go:generate git -C gguf apply ../patches/0001-remove-warm-up-logging.patch
+//go:generate cmake -S gguf -B gguf/build/metal -DLLAMA_METAL=on -DLLAMA_ACCELERATE=on -DLLAMA_K_QUANTS=on -DCMAKE_SYSTEM_PROCESSOR=arm64 -DCMAKE_OSX_ARCHITECTURES=arm64 -DCMAKE_OSX_DEPLOYMENT_TARGET=11.0
+//go:generate cmake --build gguf/build/metal --target server --config Release
--- a/llm/llama.cpp/generate_linux.go
+++ b/llm/llama.cpp/generate_linux.go
@@ -0,0 +1,22 @@
+package llm
+
+//go:generate git submodule init
+
+//go:generate git submodule update --force ggml
+//go:generate git -C ggml apply ../patches/0001-add-detokenize-endpoint.patch
+//go:generate git -C ggml apply ../patches/0002-34B-model-support.patch
+//go:generate git -C ggml apply ../patches/0005-ggml-support-CUDA-s-half-type-for-aarch64-1455-2670.patch
+//go:generate git -C ggml apply ../patches/0001-copy-cuda-runtime-libraries.patch
+//go:generate cmake -S ggml -B ggml/build/cpu -DLLAMA_K_QUANTS=on
+//go:generate cmake --build ggml/build/cpu --target server --config Release
+
+//go:generate git submodule update --force gguf
+//go:generate git -C gguf apply ../patches/0001-copy-cuda-runtime-libraries.patch
+//go:generate git -C gguf apply ../patches/0001-remove-warm-up-logging.patch
+//go:generate cmake -S gguf -B gguf/build/cpu -DLLAMA_K_QUANTS=on
+//go:generate cmake --build gguf/build/cpu --target server --config Release
+
+//go:generate cmake -S ggml -B ggml/build/cuda -DLLAMA_CUBLAS=on -DLLAMA_ACCELERATE=on -DLLAMA_K_QUANTS=on
+//go:generate cmake --build ggml/build/cuda --target server --config Release
+//go:generate cmake -S gguf -B gguf/build/cuda -DLLAMA_CUBLAS=on -DLLAMA_ACCELERATE=on -DLLAMA_K_QUANTS=on
+//go:generate cmake --build gguf/build/cuda --target server --config Release
--- a/llm/llama.cpp/generate_windows.go
+++ b/llm/llama.cpp/generate_windows.go
@@ -0,0 +1,14 @@
+package llm
+
+//go:generate git submodule init
+
+//go:generate git submodule update --force ggml
+//go:generate git -C ggml apply ../patches/0001-add-detokenize-endpoint.patch
+//go:generate git -C ggml apply ../patches/0002-34B-model-support.patch
+//go:generate cmake -S ggml -B ggml/build/cpu -DLLAMA_K_QUANTS=on
+//go:generate cmake --build ggml/build/cpu --target server --config Release
+
+//go:generate git submodule update --force gguf
+//go:generate git -C gguf apply ../patches/0001-remove-warm-up-logging.patch
+//go:generate cmake -S gguf -B gguf/build/cpu -DLLAMA_K_QUANTS=on
+//go:generate cmake --build gguf/build/cpu --target server --config Release
--- a/Show More
+++ b/Show More