diff --git a/discover/llama_server.go b/discover/llama_server.go index a66c34e87..ba8ace58c 100644 --- a/discover/llama_server.go +++ b/discover/llama_server.go @@ -199,13 +199,9 @@ var ( cudaRuntimeDirRegex = regexp.MustCompile(`^cuda_v(\d+)$`) ) -// parseLlamaServerDevices parses the combined output of llama-server discovery. -// It extracts device info, ROCm gfx targets, CUDA compute capabilities, and -// CUDA compiled architecture lists. -func parseLlamaServerDevices(output string, libDirs []string) []ml.DeviceInfo { - return parseLlamaServerDevicesWithNative(output, "", libDirs, nil) -} - +// parseLlamaServerDevicesWithNative parses the combined output of llama-server +// discovery. It extracts device info, ROCm gfx targets, CUDA compute +// capabilities, and CUDA compiled architecture lists. func parseLlamaServerDevicesWithNative(output, nativeOutput string, libDirs []string, nativeDevices []nativeProbeDevice) []ml.DeviceInfo { combined := output if nativeOutput != "" { diff --git a/discover/llama_server_test.go b/discover/llama_server_test.go index 563da94e1..0418dc2fb 100644 --- a/discover/llama_server_test.go +++ b/discover/llama_server_test.go @@ -243,7 +243,7 @@ Available devices: if tt.libDirs == nil { tt.libDirs = []string{"/lib/ollama"} } - devices := parseLlamaServerDevices(tt.output, tt.libDirs) + devices := parseLlamaServerDevicesWithNative(tt.output, "", tt.libDirs, nil) if len(devices) != len(tt.want) { t.Fatalf("got %d devices, want %d", len(devices), len(tt.want)) } @@ -357,7 +357,7 @@ Available devices: libDirs = []string{"/lib/ollama"} } - got := parseLlamaServerDevices(tt.output, libDirs) + got := parseLlamaServerDevicesWithNative(tt.output, "", libDirs, nil) if len(got) != len(tt.want) { t.Fatalf("got %d devices, want %d", len(got), len(tt.want)) } diff --git a/ml/backend.go b/ml/backend.go index 37efd365e..7c3340edd 100644 --- a/ml/backend.go +++ b/ml/backend.go @@ -19,9 +19,6 @@ type Backend interface { Load(ctx context.Context, progress func(float32)) error - // BackendMemory returns the memory allocations that were made for this model - BackendMemory() BackendMemory - Config() fs.Config Get(name string) Tensor NewContext() Context @@ -66,9 +63,6 @@ type BackendParams struct { // NumThreads sets the number of threads to use if running on the CPU NumThreads int - // GPULayers is the set of layers to offload to GPUs - GPULayers GPULayersList - // FlashAttention indicates that we should use a fused flash attention kernel FlashAttention FlashAttentionType } diff --git a/ml/device.go b/ml/device.go index fbeb07139..408f1cd7d 100644 --- a/ml/device.go +++ b/ml/device.go @@ -2,133 +2,17 @@ package ml import ( "context" - "encoding/binary" - "encoding/json" "fmt" - "hash/maphash" - "io" "log/slog" - "math" - "net/http" "os" "runtime" - "slices" "sort" "strconv" "strings" - "time" "github.com/ollama/ollama/format" - "github.com/ollama/ollama/logutil" ) -// GPULayers is a set of layers to be allocated on a single GPU -type GPULayers struct { - DeviceID - - // Layers is a set of layer indicies to load - Layers []int -} - -// FirstLayer returns the smallest layer index scheduled on this GPU, or MaxInt when empty. -func (g GPULayers) FirstLayer() int { - if len(g.Layers) == 0 { - return math.MaxInt - } - - first := g.Layers[0] - for i := 1; i < len(g.Layers); i++ { - if g.Layers[i] < first { - first = g.Layers[i] - } - } - - return first -} - -func (g GPULayers) String() string { - if len(g.Layers) == 0 { - return "" - } - - slices.Sort(g.Layers) - - contiguous := true - base := g.Layers[0] - for i := range g.Layers { - if g.Layers[i] != base+i { - contiguous = false - break - } - } - - if contiguous { - return fmt.Sprintf("ID:%v Layers:%v(%v..%v)", g.ID, len(g.Layers), g.Layers[0], g.Layers[len(g.Layers)-1]) - } else { - return fmt.Sprintf("ID:%v Layers:%v%v", g.ID, len(g.Layers), g.Layers) - } -} - -// GPULayersList is a set of layer allocations across multiple GPUs -type GPULayersList []GPULayers - -func (l GPULayersList) Len() int { return len(l) } -func (l GPULayersList) Swap(i, j int) { l[i], l[j] = l[j], l[i] } - -// Sort by the ordering of the layers offloaded -func (l GPULayersList) Less(i, j int) bool { - li := l[i].FirstLayer() - lj := l[j].FirstLayer() - - return li < lj -} - -func (l GPULayersList) String() string { - if l.Sum() > 0 { - return fmt.Sprintf("%v%v", l.Sum(), []GPULayers(l)) - } else { - return fmt.Sprintf("%v", []GPULayers(l)) - } -} - -// Sum is the total number of layers assigned across all GPUs -func (l GPULayersList) Sum() int { - var sum int - - for _, g := range l { - sum += len(g.Layers) - } - - return sum -} - -var h maphash.Hash - -// Hash is an identifier of this layer assignment -func (l GPULayersList) Hash() uint64 { - h.Reset() - for _, g := range l { - if len(g.Layers) > 0 { - h.WriteString(g.ID + g.Library) - for _, l := range g.Layers { - binary.Write(&h, binary.NativeEndian, int64(l)) - } - } - } - - return h.Sum64() -} - -// ErrNoMem is returned when panicing due to insufficient memory. It includes -// the attempted memory allocation. -type ErrNoMem struct { - BackendMemory -} - -func (e ErrNoMem) Error() string { - return fmt.Sprintf("insufficient memory - required allocations: %+v", e.BackendMemory) -} - // Minimal unique device identification type DeviceID struct { // ID is an identifier for the device for matching with system @@ -142,138 +26,6 @@ type DeviceID struct { Library string `json:"backend,omitempty"` } -// DeviceMemory provides a breakdown of the memory needed -// per device, such as a CPU or GPU. -type DeviceMemory struct { - DeviceID - - // Name is the name of the device as labeled by the backend. It - // may not be persistent across instances of the runner. - Name string - - // Weights is the per-layer memory needed for the model weights. - Weights []uint64 - - // Cache is the per-layer memory needed for the KV cache. - Cache []uint64 - - // Graph is the size of the compute graph. It is not per-layer. - Graph uint64 -} - -func sumMemory(mem []uint64) uint64 { - var sum uint64 - - for _, m := range mem { - sum += m - } - - return sum -} - -// Size returns the total size of the memory required by this device -func (m DeviceMemory) Size() uint64 { - return sumMemory(m.Weights) + sumMemory(m.Cache) + m.Graph -} - -func memoryPresent(mem []uint64) bool { - return slices.ContainsFunc(mem, func(m uint64) bool { return m != 0 }) -} - -func (m DeviceMemory) LogValue() slog.Value { - var attrs []slog.Attr - if memoryPresent(m.Weights) { - attrs = append(attrs, slog.Any("Weights", m.Weights)) - } - - if memoryPresent(m.Cache) { - attrs = append(attrs, slog.Any("Cache", m.Cache)) - } - - if m.Graph != 0 { - attrs = append(attrs, slog.Any("Graph", m.Graph)) - } - - if len(attrs) > 0 && m.ID != "" { - attrs = append([]slog.Attr{slog.String("ID", m.ID)}, attrs...) - } - - return slog.GroupValue(attrs...) -} - -// BackendMemory provides the amount of memory required to load the model -// per device based on the BackendParams. In some cases, not all required -// allocations will be known at this point. However, the size of the most recent -// allocation is guaranteed to be provided so that if it failed, the caller can -// accommodate that to make forward progress. -type BackendMemory struct { - // InputWeights are always located on the CPU and cannot be moved - InputWeights uint64 - - // CPU model components are located in system memory. This does not - // include unified memory allocated through the GPU. - CPU DeviceMemory - - // GPU model components are located on one or more GPUs. - GPUs []DeviceMemory -} - -func (m BackendMemory) LogValue() slog.Value { - var attrs []slog.Attr - if m.InputWeights != 0 { - attrs = append(attrs, slog.Any("InputWeights", m.InputWeights)) - } - - attrs = append(attrs, slog.Any(m.CPU.Name, m.CPU)) - for _, g := range m.GPUs { - attrs = append(attrs, slog.Any(g.Name, g)) - } - - return slog.GroupValue(attrs...) -} - -// Log prints a high level summary of the memory -func (m BackendMemory) Log(level slog.Level) { - var total uint64 - - for _, gpu := range m.GPUs { - if sum := sumMemory(gpu.Weights); sum > 0 { - slog.Log(context.TODO(), level, "model weights", "device", gpu.Name, "size", format.HumanBytes2(sum)) - total += sum - } - } - if sum := m.InputWeights + sumMemory(m.CPU.Weights); sum > 0 { - slog.Log(context.TODO(), level, "model weights", "device", m.CPU.Name, "size", format.HumanBytes2(sum)) - total += sum - } - - for _, gpu := range m.GPUs { - if sum := sumMemory(gpu.Cache); sum > 0 { - slog.Log(context.TODO(), level, "kv cache", "device", gpu.Name, "size", format.HumanBytes2(sum)) - total += sum - } - } - if sum := sumMemory(m.CPU.Cache); sum > 0 { - slog.Log(context.TODO(), level, "kv cache", "device", m.CPU.Name, "size", format.HumanBytes2(sum)) - total += sum - } - - for _, gpu := range m.GPUs { - if sum := gpu.Graph; sum > 0 { - slog.Log(context.TODO(), level, "compute graph", "device", gpu.Name, "size", format.HumanBytes2(sum)) - total += sum - } - } - if sum := m.CPU.Graph; sum > 0 { - slog.Log(context.TODO(), level, "compute graph", "device", m.CPU.Name, "size", format.HumanBytes2(sum)) - total += sum - } - - if total > 0 { - slog.Log(context.TODO(), level, "total memory", "size", format.HumanBytes2(total)) - } -} - type DeviceInfo struct { DeviceID @@ -378,28 +130,6 @@ func (a ByFreeMemory) Less(i, j int) bool { return a[i].FreeMemory < a[j].FreeMemory } -// ByPerformance groups devices by similar speed -func ByPerformance(l []DeviceInfo) [][]DeviceInfo { - resp := [][]DeviceInfo{} - scores := []bool{} - for _, info := range l { - found := false - requested := info.Integrated - for i, score := range scores { - if score == requested { - resp[i] = append(resp[i], info) - found = true - break - } - } - if !found { - scores = append(scores, requested) - resp = append(resp, []DeviceInfo{info}) - } - } - return resp -} - func ByLibrary(l []DeviceInfo) [][]DeviceInfo { resp := [][]DeviceInfo{} libs := []string{} @@ -810,51 +540,3 @@ type FilteredRunnerDiscovery interface { GetActiveDeviceIDs() []DeviceID } -func GetDevicesFromRunner(ctx context.Context, runner BaseRunner) ([]DeviceInfo, error) { - var moreDevices []DeviceInfo - port := runner.GetPort() - tick := time.Tick(10 * time.Millisecond) - for { - select { - case <-ctx.Done(): - return nil, fmt.Errorf("failed to finish discovery before timeout") - case <-tick: - r, err := http.NewRequestWithContext(ctx, http.MethodGet, fmt.Sprintf("http://127.0.0.1:%d/info", port), nil) - if err != nil { - return nil, fmt.Errorf("failed to create request: %w", err) - } - r.Header.Set("Content-Type", "application/json") - - resp, err := http.DefaultClient.Do(r) - if err != nil { - // slog.Warn("failed to send request", "error", err) - if runner.HasExited() { - return nil, fmt.Errorf("runner crashed") - } - continue - } - defer resp.Body.Close() - - if resp.StatusCode == http.StatusNotFound { - // old runner, fall back to bootstrapping model - return nil, fmt.Errorf("llamarunner free vram reporting not supported") - } - - body, err := io.ReadAll(resp.Body) - if err != nil { - slog.Warn("failed to read response", "error", err) - continue - } - if resp.StatusCode != 200 { - logutil.Trace("runner failed to discover free VRAM", "status", resp.StatusCode, "response", body) - return nil, fmt.Errorf("runner error: %s", string(body)) - } - - if err := json.Unmarshal(body, &moreDevices); err != nil { - slog.Warn("unmarshal encode response", "error", err) - continue - } - return moreDevices, nil - } - } -}