Runner for Ollama engine

This provides integration with the new Ollama engine (5824541 next ollama runner (#7913)) and the rest of the Ollama infrastructure such as the runner and Ollama server. In addition, it also builds out the KV cache infrastructure to support requirements of how Ollama runs models such as: - Parallel processing - Memory management for defragmentation and shifting - Multi-modal modals Both old and new engines continue to be supported. By default, only the old engine is used. To enable the new engine: Start the server with the OLLAMA_NEW_ENGINE environment variable set: OLLAMA_NEW_ENGINE=1 ./ollama serve Start a model that is supported by the Ollama engine. This one is Llama 3.1 8b Q4_K_M: ./ollama run jessegross/llama3.1
2025-05-11 18:36:41 +02:00 · 2024-12-17 19:59:41 -08:00 · 2024-12-17 19:59:41 -08:00 · ed443a0393
commit ed443a0393
parent 6945617af5
31 changed files with 2952 additions and 244 deletions
--- a/model/model.go
+++ b/model/model.go
@ -1,6 +1,7 @@
 package model

 import (
+	"errors"
 	"fmt"
 	"image"
 	_ "image/jpeg"
@ -15,102 +16,42 @@ import (
 	_ "golang.org/x/image/tiff"
 	_ "golang.org/x/image/webp"

-	"github.com/ollama/ollama/cache"
+	"github.com/ollama/ollama/kvcache"
 	"github.com/ollama/ollama/ml"
 	_ "github.com/ollama/ollama/ml/backend"
 )

-type Cache struct {
-	cache.Cache
-	cache.Options
-}
-
-func (c Cache) Sub(i int) Cache {
-	if c.Cache != nil {
-		return Cache{
-			Cache:   c.Cache.Sub(i),
-			Options: c.Options,
-		}
-	}
-
-	return c
-}
-
-func (c Cache) Put(ctx ml.Context, key, value ml.Tensor, opts cache.Options) (ml.Tensor, ml.Tensor) {
-	if c.Cache != nil {
-		return c.Cache.Put(ctx, key, value, opts)
-	}
-
-	return key, value
-}
-
 type Options struct {
-	inputs []int32
-
-	Offset int
+	Inputs    []int32
+	Positions []int32
+	Sequences []int
+	Outputs   []int32

 	Images []image.Image
-
-	Cache
 }

-func (opts Options) Inputs() []int32 {
-	return opts.inputs[opts.Offset:]
-}
-
-func (opts Options) Positions() []int32 {
-	positions := make([]int32, len(opts.inputs)-opts.Offset)
-	for i := range positions {
-		positions[i] = int32(opts.Offset + i)
-	}
-
-	return positions
-}
-
-type OptionsFunc func(Model, *Options)
-
-func WithInputIDs(ids []int32) OptionsFunc {
-	return func(m Model, opts *Options) {
-		opts.inputs = ids
-	}
-}
-
-func WithOffset(offset int) OptionsFunc {
-	return func(m Model, opts *Options) {
-		opts.Offset = offset
-		opts.Cache.Position = offset
-	}
-}
-
-func WithImage(img image.Image) OptionsFunc {
-	return func(m Model, opts *Options) {
-		opts.Images = append(opts.Images, img)
-	}
-}
-
-func WithCache(c cache.Cache) OptionsFunc {
-	return func(m Model, opts *Options) {
-		opts.Cache = Cache{
-			Cache: c,
-			Options: cache.Options{
-				Position: opts.Offset,
-			},
-		}
-	}
+type config struct {
+	Cache kvcache.Cache
 }

 type Base struct {
 	b ml.Backend
+	config
 }

 func (m *Base) Backend() ml.Backend {
 	return m.b
 }

+func (m *Base) Config() config {
+	return m.config
+}
+
 type Model interface {
 	Forward(ml.Context, Options) (ml.Tensor, error)

 	Backend() ml.Backend
+	Config() config
 }

 var models = make(map[string]func(ml.Config) (Model, error))
@ -146,12 +87,14 @@ func New(s string) (Model, error) {
 		return nil, err
 	}

+	base := Base{b: b, config: m.Config()}
+
 	v := reflect.ValueOf(m)
-	v.Elem().Set(populateFields(b, v.Elem()))
+	v.Elem().Set(populateFields(base, v.Elem()))
 	return m, nil
 }

-func populateFields(b ml.Backend, v reflect.Value, tags ...Tag) reflect.Value {
+func populateFields(base Base, v reflect.Value, tags ...Tag) reflect.Value {
 	t := v.Type()

 	if t.Kind() == reflect.Struct {
@ -170,7 +113,7 @@ func populateFields(b ml.Backend, v reflect.Value, tags ...Tag) reflect.Value {
 			}

 			if tt == reflect.TypeOf((*Base)(nil)).Elem() {
-				vv.Set(reflect.ValueOf(Base{b: b}))
+				vv.Set(reflect.ValueOf(base))
 			} else if tt == reflect.TypeOf((*ml.Tensor)(nil)).Elem() {
 				var fn func([]Tag) [][]string
 				fn = func(tags []Tag) (values [][]string) {
@ -196,21 +139,21 @@ func populateFields(b ml.Backend, v reflect.Value, tags ...Tag) reflect.Value {

 				names := fn(tagsCopy)
 				for _, name := range names {
-					if tensor := b.Get(strings.Join(name, ".")); tensor != nil {
+					if tensor := base.Backend().Get(strings.Join(name, ".")); tensor != nil {
 						slog.Debug("found tensor", "", tensor)
 						vv.Set(reflect.ValueOf(tensor))
 						break
 					}
 				}
 			} else if tt.Kind() == reflect.Pointer || tt.Kind() == reflect.Interface {
-				setPointer(b, vv, tagsCopy)
+				setPointer(base, vv, tagsCopy)
 			} else if tt.Kind() == reflect.Slice || tt.Kind() == reflect.Array {
 				for i := range vv.Len() {
 					vvv := vv.Index(i)
 					if vvv.Kind() == reflect.Pointer || vvv.Kind() == reflect.Interface {
-						setPointer(b, vvv, append(tagsCopy, Tag{Name: strconv.Itoa(i)}))
+						setPointer(base, vvv, append(tagsCopy, Tag{Name: strconv.Itoa(i)}))
 					} else {
-						vvv.Set(populateFields(b, vvv, append(tagsCopy, Tag{Name: strconv.Itoa(i)})...))
+						vvv.Set(populateFields(base, vvv, append(tagsCopy, Tag{Name: strconv.Itoa(i)})...))
 					}
 				}
 			}
@ -228,7 +171,7 @@ func populateFields(b ml.Backend, v reflect.Value, tags ...Tag) reflect.Value {
 	return v
 }

-func setPointer(b ml.Backend, v reflect.Value, tags []Tag) {
+func setPointer(base Base, v reflect.Value, tags []Tag) {
 	vv := v
 	if v.Kind() == reflect.Interface {
 		if v.IsNil() {
@ -243,7 +186,7 @@ func setPointer(b ml.Backend, v reflect.Value, tags []Tag) {
 		vv = reflect.New(v.Type().Elem()).Elem()
 	}

-	if f := populateFields(b, vv, tags...); f.CanAddr() {
+	if f := populateFields(base, vv, tags...); f.CanAddr() {
 		v.Set(f.Addr())
 	}
 }
@ -277,18 +220,27 @@ func canNil(t reflect.Type) bool {
 		t.Kind() == reflect.Slice
 }

-func Forward(m Model, optsFuncs ...OptionsFunc) (ml.Tensor, error) {
-	var opts Options
-	for _, optsFunc := range optsFuncs {
-		optsFunc(m, &opts)
+func Forward(ctx ml.Context, m Model, opts Options) (ml.Tensor, error) {
+	if len(opts.Positions) != len(opts.Sequences) {
+		return nil, fmt.Errorf("length of positions (%v) must match length of seqs (%v)", len(opts.Positions), len(opts.Sequences))
+	}
+
+	if len(opts.Positions) < 1 {
+		return nil, errors.New("batch size cannot be less than 1")
+	}
+
+	cache := m.Config().Cache
+	if cache != nil {
+		err := cache.StartForward(ctx, opts.Positions, opts.Sequences)
+		if err != nil {
+			return nil, err
+		}
 	}

-	ctx := m.Backend().NewContext()
 	t, err := m.Forward(ctx, opts)
 	if err != nil {
 		return nil, err
 	}
-	defer ctx.Close()

 	ctx.Forward(t)
 	ctx.Compute(t)