@ -10,7 +10,6 @@ import "C"
import (
"context"
"errors"
"fmt"
"io"
"log/slog"
@ -66,6 +65,12 @@ type Backend struct {
// layers is the backend used for repeating layers
layers map [ int ] * C . struct_ggml_backend_buffer_type
// requiredMemory is the cumulative memory allocations needed by the backend
requiredMemory * ml . BackendMemory
// btDeviceMemory maps from a buffer type to the memory allocations associated with that device
btDeviceMemory map [ * C . struct_ggml_backend_buffer_type ] * ml . DeviceMemory
flashAttention bool
// maxGraphNodes is the maximum allowed number of graph nodes in this scheduler
@ -94,6 +99,9 @@ func New(modelPath string, params ml.BackendParams) (ml.Backend, error) {
"num_key_values" , len ( meta . KV ( ) ) ,
)
var requiredMemory ml . BackendMemory
btDeviceMemory := make ( map [ * C . struct_ggml_backend_buffer_type ] * ml . DeviceMemory )
type deviceBufferType struct {
d * C . struct_ggml_backend_device
bts [ ] * C . struct_ggml_backend_buffer_type
@ -114,6 +122,8 @@ func New(modelPath string, params ml.BackendParams) (ml.Backend, error) {
}
}
blocks := int ( meta . KV ( ) . BlockCount ( ) )
// create list of buffer types for the cpu
cpuDeviceBufferType := deviceBufferType { d : C . ggml_backend_dev_by_type ( C . GGML_BACKEND_DEVICE_TYPE_CPU ) }
for _ , d := range append ( accels , append ( gpus , cpus ... ) ... ) {
@ -121,17 +131,27 @@ func New(modelPath string, params ml.BackendParams) (ml.Backend, error) {
case C . GGML_BACKEND_DEVICE_TYPE_CPU ,
C . GGML_BACKEND_DEVICE_TYPE_ACCEL :
cpuDeviceBufferType . bts = append ( cpuDeviceBufferType . bts , C . ggml_backend_dev_buffer_type ( d ) )
btDeviceMemory [ C . ggml_backend_dev_buffer_type ( d ) ] = & requiredMemory . CPU
}
}
requiredMemory . CPU . Name = C . GoString ( C . ggml_backend_dev_name ( cpuDeviceBufferType . d ) )
requiredMemory . CPU . Weights = make ( [ ] ml . Memory , blocks + 1 )
requiredMemory . CPU . Cache = make ( [ ] ml . Memory , blocks + 1 )
// create list of buffer types for each gpu
var gpuDeviceBufferTypes [ ] deviceBufferType
for _ , d := range gpus {
requiredMemory . GPUs = make ( [ ] ml . DeviceMemory , len ( gpus ) )
for i , d := range gpus {
bt := C . ggml_backend_dev_buffer_type ( d )
gpuDeviceBufferTypes = append ( gpuDeviceBufferTypes , deviceBufferType {
d : d ,
bts : append ( [ ] * C . struct_ggml_backend_buffer_type { bt } , cpuDeviceBufferType . bts ... ) ,
} )
btDeviceMemory [ bt ] = & requiredMemory . GPUs [ i ]
requiredMemory . GPUs [ i ] . Name = C . GoString ( C . ggml_backend_dev_name ( d ) )
requiredMemory . GPUs [ i ] . Weights = make ( [ ] ml . Memory , blocks + 1 )
requiredMemory . GPUs [ i ] . Cache = make ( [ ] ml . Memory , blocks + 1 )
}
useDefaultSplit := true
@ -170,8 +190,6 @@ func New(modelPath string, params ml.BackendParams) (ml.Backend, error) {
// inputs always use cpu
input := cpuDeviceBufferType
blocks := int ( meta . KV ( ) . BlockCount ( ) )
// define a range of gpu layers. anything outside of this range is assigned to the cpu
gpuRangeStart := max ( 0 , blocks - params . NumGPULayers )
gpuRangeStop := min ( gpuRangeStart + params . NumGPULayers , blocks + 1 )
@ -212,7 +230,7 @@ func New(modelPath string, params ml.BackendParams) (ml.Backend, error) {
// contexts are shared by tensors of the same buffer type
ctxs := make ( map [ * C . struct_ggml_backend_buffer_type ] * C . struct_ggml_context )
createTensor := func ( t tensor , bts [ ] * C . struct_ggml_backend_buffer_type ) * C . struct_ggml_tensor {
createTensor := func ( t tensor , bts [ ] * C . struct_ggml_backend_buffer_type , layer int ) * C . struct_ggml_tensor {
for _ , bt := range bts {
if _ , ok := ctxs [ bt ] ; ! ok {
ctxs [ bt ] = C . ggml_init ( C . struct_ggml_init_params {
@ -238,6 +256,16 @@ func New(modelPath string, params ml.BackendParams) (ml.Backend, error) {
C . ggml_set_name ( tt , cname )
slog . Log ( context . TODO ( ) , logutil . LevelTrace , "created tensor" , "name" , name , "shape" , t . source . Shape , "dtype" , t . source . Kind , "buffer_type" , C . GoString ( C . ggml_backend_buft_name ( bt ) ) )
size := pad ( C . ggml_backend_buft_get_alloc_size ( bt , tt ) , C . ggml_backend_buft_get_alignment ( bt ) )
if layer == - 1 {
// Assume that InputWeights can be allocated - they're always in system memory and can't be moved in any case
requiredMemory . InputWeights . Status = ml . Allocated
requiredMemory . InputWeights . Size += uint64 ( size )
} else {
btDeviceMemory [ bt ] . Weights [ layer ] . Size += uint64 ( size )
}
//nolint:staticcheck // TODO: check if buffer type supports this tensor
return tt
}
@ -259,22 +287,22 @@ func New(modelPath string, params ml.BackendParams) (ml.Backend, error) {
for _ , t := range meta . Tensors ( ) . Items ( ) {
switch {
case contains ( t . Name , "position_embd" , "token_embd" , "token_norm_embd" , "token_types" ) :
createTensor ( tensor { source : t } , input . bts )
createTensor ( tensor { source : t } , input . bts , - 1 )
if _ , ok := meta . Tensors ( ) . GroupLayers ( ) [ "output" ] ; ! ok && t . Name == "token_embd.weight" {
createTensor ( tensor { source : t , target : "output.weight" } , output . bts )
createTensor ( tensor { source : t , target : "output.weight" } , output . bts , blocks )
}
case contains ( t . Name , "cls" , "output" , "output_norm" ) :
createTensor ( tensor { source : t } , output . bts )
createTensor ( tensor { source : t } , output . bts , blocks )
case strings . HasPrefix ( t . Name , "v." ) || strings . HasPrefix ( t . Name , "mm." ) :
// TODO: assign vision tensors to the gpu if possible
createTensor ( tensor { source : t } , output . bts )
createTensor ( tensor { source : t } , output . bts , blocks )
case contains ( t . Name , "rope_freqs" , "rope_factors_long" , "rope_factors_short" ) :
// these tensors should be repeated per layer
for i , layer := range layers {
createTensor ( tensor {
source : t ,
target : "blk." + strconv . Itoa ( i ) + "." + t . Name ,
} , layer . bts )
} , layer . bts , i )
}
default :
layerIndex := - 1
@ -285,10 +313,10 @@ func New(modelPath string, params ml.BackendParams) (ml.Backend, error) {
}
if layerIndex >= 0 {
createTensor ( tensor { source : t } , layers [ layerIndex ] . bts )
createTensor ( tensor { source : t } , layers [ layerIndex ] . bts , layerIndex )
} else {
// load all other tensors on the cpu
createTensor ( tensor { source : t } , input . bts )
createTensor ( tensor { source : t } , input . bts , - 1 )
}
}
}
@ -301,8 +329,18 @@ func New(modelPath string, params ml.BackendParams) (ml.Backend, error) {
}
b := C . ggml_backend_alloc_ctx_tensors_from_buft ( c , bt )
for i := range btDeviceMemory [ bt ] . Weights {
if btDeviceMemory [ bt ] . Weights [ i ] . Size != 0 {
if b != nil {
btDeviceMemory [ bt ] . Weights [ i ] . Status = ml . Allocated
} else {
btDeviceMemory [ bt ] . Weights [ i ] . Status = ml . Failed
}
}
}
if b == nil {
return nil , fmt . Errorf ( "unable to allocate memory from device %v for model weights" , C . GoString ( C . ggml_backend_buft_name ( bt ) ) )
panic ( ml . ErrNoMem { BackendMemory : requiredMemory } )
}
C . ggml_backend_buffer_set_usage ( b , C . GGML_BACKEND_BUFFER_USAGE_WEIGHTS )
@ -367,7 +405,9 @@ func New(modelPath string, params ml.BackendParams) (ml.Backend, error) {
}
return m
} ( ) ,
maxGraphNodes : maxGraphNodes ,
requiredMemory : & requiredMemory ,
btDeviceMemory : btDeviceMemory ,
maxGraphNodes : maxGraphNodes ,
} , nil
}
@ -446,6 +486,10 @@ func (b *Backend) Load(ctx context.Context, progress func(float32)) error {
return nil
}
func ( b * Backend ) BackendMemory ( ) ml . BackendMemory {
return * b . requiredMemory
}
func ( b * Backend ) Config ( ) fs . Config {
return b . meta . KV ( )
}
@ -477,6 +521,7 @@ func (b *Backend) NewContextSize(n int) ml.Context {
no_alloc : true ,
} ) ,
allocatedBuffers : & allocatedBuffers ,
layer : - 1 ,
}
}
@ -503,6 +548,9 @@ type Context struct {
// maxGraphNodes is the maximum allowed number of graph nodes in this context
maxGraphNodes int
// layer is the graph layer that this context is allocating for - assumed to be cache
layer int
}
func ( c * Context ) Input ( ) ml . Context {
@ -513,6 +561,7 @@ func (c *Context) Input() ml.Context {
buft : c . b . input ,
allocatedBuffers : c . allocatedBuffers ,
maxGraphNodes : c . maxGraphNodes ,
layer : - 1 ,
}
}
@ -527,6 +576,7 @@ func (c *Context) Layer(i int) ml.Context {
buft : buft ,
allocatedBuffers : c . allocatedBuffers ,
maxGraphNodes : c . maxGraphNodes ,
layer : i ,
}
}
@ -564,22 +614,34 @@ func (c *Context) Compute(tensors ...ml.Tensor) {
}
}
func ( c * Context ) Reserve ( ) error {
if ! C . ggml_backend_sched_reserve ( c . b . sched , c . graph ) {
C . ggml_backend_sched_reset ( c . b . sched )
return errors . New ( "failed to reserve graph" )
}
func ( c * Context ) Reserve ( ) {
reserved := C . ggml_backend_sched_reserve ( c . b . sched , c . graph )
slog . Debug ( "compute graph" , "nodes" , C . ggml_graph_n_nodes ( c . graph ) , "splits" , C . ggml_backend_sched_get_n_splits ( c . b . sched ) )
// Reserve may get called multiple times for different graphs - we just want the last run, which will contain the max allocations
for _ , bt := range c . b . schedBufts {
c . b . btDeviceMemory [ bt ] . Graph = ml . Memory { }
}
for i := range c . b . schedBackends {
size := C . ggml_backend_sched_get_buffer_size ( c . b . sched , c . b . schedBackends [ i ] )
bufferStatus := C . ggml_backend_sched_get_attempted_buffer_size ( c . b . sched , c . b . schedBackends [ i ] )
graph := & c . b . btDeviceMemory [ c . b . schedBufts [ i ] ] . Graph
graph . Size += uint64 ( bufferStatus . size )
if bufferStatus . allocated && graph . Status != ml . Failed {
graph . Status = ml . Allocated
} else {
graph . Status = ml . Failed
}
slog . Info ( "compute graph" , "backend" , C . GoString ( C . ggml_backend_name ( c . b . schedBackends [ i ] ) ) , "buffer_type" , C . GoString ( C . ggml_backend_buft_name ( c . b . schedBufts [ i ] ) ) ,
"size" , format . HumanBytes2 ( uint64 ( size ) ) )
"size" , format . HumanBytes2 ( uint64 ( bufferStatus . size ) ) )
}
C . ggml_backend_sched_reset ( c . b . sched )
return nil
if ! reserved {
panic ( ml . ErrNoMem { BackendMemory : * c . b . requiredMemory } )
}
}
func ( c * Context ) MaxGraphNodes ( ) int {
@ -599,7 +661,7 @@ func pad(length, pad C.size_t) C.size_t {
return ( ( length + pad - 1 ) / pad ) * pad
}
func ( c * Context ) newTensor ( dtype ml . DType , shape [ ] int ) ( ml . Tensor , error ) {
func ( c * Context ) newTensor ( dtype ml . DType , shape [ ] int ) ml . Tensor {
if c . buft == nil {
panic ( "set Input or Layer before creating tensors" )
}
@ -622,7 +684,7 @@ func (c *Context) newTensor(dtype ml.DType, shape []int) (ml.Tensor, error) {
if len ( shape ) < 1 || shape [ 0 ] == 0 {
var shape C . int64_t = 0
return & Tensor { b : c . b , t : C . ggml_new_tensor ( c . ctx , cdtype , 1 , & shape ) } , nil
return & Tensor { b : c . b , t : C . ggml_new_tensor ( c . ctx , cdtype , 1 , & shape ) }
} else if len ( shape ) > 4 {
panic ( "unsupported number of dimensions" )
}
@ -635,31 +697,34 @@ func (c *Context) newTensor(dtype ml.DType, shape []int) (ml.Tensor, error) {
t := C . ggml_new_tensor ( c . ctx , cdtype , C . int ( len ( shape ) ) , shapeToGGML ( shape ) )
size := pad ( C . ggml_backend_buft_get_alloc_size ( c . buft , t ) , C . ggml_backend_buft_get_alignment ( c . buft ) )
b := C . ggml_backend_buft_alloc_buffer ( c . buft , size )
if c . layer >= 0 {
cache := & c . b . btDeviceMemory [ c . buft ] . Cache [ c . layer ]
cache . Size += uint64 ( size )
if b != nil {
cache . Status = ml . Allocated
} else {
cache . Status = ml . Failed
}
}
if b == nil {
return nil , fmt . Errorf ( "unable to allocate %v from device %v for new tensor" , format . HumanBytes2 ( uint64 ( size ) ) , C . GoString ( C . ggml_backend_buft_name ( c . buft ) ) )
panic ( ml . ErrNoMem { BackendMemory : * c . b . requiredMemory } )
}
* c . allocatedBuffers = append ( * c . allocatedBuffers , b )
* c . allocatedBuffers = append ( * c . allocatedBuffers , b )
C . ggml_backend_tensor_alloc ( b , t , C . ggml_backend_buffer_get_base ( b ) )
return & Tensor { b : c . b , t : t } , nil
return & Tensor { b : c . b , t : t }
}
func ( c * Context ) Empty ( dtype ml . DType , shape ... int ) ml . Tensor {
t , err := c . newTensor ( dtype , shape )
if err != nil {
panic ( err )
}
return t
return c . newTensor ( dtype , shape )
}
func ( c * Context ) Zeros ( dtype ml . DType , shape ... int ) ml . Tensor {
t , err := c . newTensor ( dtype , shape )
if err != nil {
panic ( err )
}
t := c . newTensor ( dtype , shape )
C . ggml_set_zero ( t . ( * Tensor ) . t )
return t
}
@ -687,10 +752,7 @@ func (c *Context) FromFloatSlice(s []float32, shape ...int) (ml.Tensor, error) {
return nil , err
}
t , err := c . newTensor ( ml . DTypeF32 , shape )
if err != nil {
return nil , err
}
t := c . newTensor ( ml . DTypeF32 , shape )
if len ( s ) > 0 {
C . ggml_backend_tensor_set ( t . ( * Tensor ) . t , unsafe . Pointer ( & s [ 0 ] ) , 0 , C . ggml_nbytes ( t . ( * Tensor ) . t ) )
@ -704,10 +766,7 @@ func (c *Context) FromIntSlice(s []int32, shape ...int) (ml.Tensor, error) {
return nil , err
}
t , err := c . newTensor ( ml . DTypeI32 , shape )
if err != nil {
return nil , err
}
t := c . newTensor ( ml . DTypeI32 , shape )
if len ( s ) > 0 {
C . ggml_backend_tensor_set ( t . ( * Tensor ) . t , unsafe . Pointer ( & s [ 0 ] ) , 0 , C . ggml_nbytes ( t . ( * Tensor ) . t ) )