Compare commits

...
3 Commits
Author SHA1 Message Date
Dante 79e2e74eb1 CUDA: fix round issue, under MSVC the CPU and GPU agree (#30229) 2026-10-09 19:59:50 +02:00
Georgi Gerganov 8e2d31e0eb graph : reorder get_rows for embeddings (#30160)
* graph : reorder get_rows for embeddings

* cont : fix gemma4 and improve input embedding construction logic

* cont : add TODO for lora

* cont : fix raw embeddings path

* gemma4 : avoid ple cast in embeddings path
2026-10-09 20:43:05 +03:00
Aleksander GrygierandPascal baef3ed9a1 ui: Models Manager Follow-up Improvements (#30228)
* common : read a GGUF's trained context from its metadata

common_get_gguf_n_ctx_train opens only the file's metadata (no_alloc,
like common_get_decision_type) and reads <arch>.context_length, so a
caller can learn the trained context without loading the model. It
accepts both u32 and u64 values and returns 0 when the file is missing,
unreadable, invalid, or reports no context length.

Assisted-by: pi:zai-org/GLM-5.3-Flash

* server : report the trained context in the models listing

update_caps already resolves the model file offline to read its
modalities, so it now reads the trained context from the same GGUF
metadata, and GET /models reports it as context_length when it is
known. A router listing then carries the context without any Hub
request, which lets the UI sort and filter by it offline.

Assisted-by: pi:zai-org/GLM-5.3-Flash

* ui : take the trained context from the models listing

The router now reports context_length per model, so the option mapping
fills contextLength from it and the manager reads it before the Hub
record. The Context column, the context sort and the context filter
then work with the Hugging Face Hub API turned off. A browser suite
guards the sort and the search, the Hub-cache driven context filter and
the re-sort when details arrive after the sort was clicked.

Assisted-by: pi:zai-org/GLM-5.3-Flash

* ui : mark favorite models with a heart

A favorited model shows a rose heart in the selector even before its
row is hovered, and the crossed heart takes its place on hover, so
unfavoriting stays one hover away. The manager table marks its
favorited rows with the same heart after the badges and capabilities.

Assisted-by: pi:zai-org/GLM-5.3-Flash

* fix: UI text nit

* fix: UI nits

* fix: Favorite models grouping in models table

* feat: Remove sorting from Status column in Models Table

* server: read the GGUF metadata once per model

Read the decision type and the trained context in a single GGUF open,
accept only a UINT32 context length like the model loader, and reset
n_ctx_train with the other caps so a failed refresh drops it.

* fix: Post-review fixes

---------

Co-authored-by: Pascal <admin@serveurperso.com>
2026-10-09 19:28:54 +02:00
21 changed files with 453 additions and 149 deletions
+19 -16
View File
@@ -1195,7 +1195,9 @@ common_decision_type common_get_decision_type(const struct llama_model * model)
return common_decision_type_from_string(buf);
}
common_decision_type common_get_decision_type(const std::string & fname) {
common_gguf_info common_get_gguf_info(const std::string & fname) {
common_gguf_info info;
struct gguf_init_params gguf_params = {
/* .no_alloc = */ true,
/* .ctx = */ nullptr,
@@ -1203,31 +1205,32 @@ common_decision_type common_get_decision_type(const std::string & fname) {
gguf_context_ptr gguf_ctx(gguf_init_from_file(fname.c_str(), gguf_params));
if (!gguf_ctx) {
return COMMON_DECISION_TYPE_UNKNOWN; // missing or unreadable file
return info; // missing or unreadable file
}
std::string arch;
const int64_t arch_id = gguf_find_key(gguf_ctx.get(), "general.architecture");
if (arch_id < 0) {
return COMMON_DECISION_TYPE_UNKNOWN; // no architecture in the metadata
if (arch_id < 0 || gguf_get_kv_type(gguf_ctx.get(), arch_id) != GGUF_TYPE_STRING) {
return info; // no architecture in the metadata
}
if (gguf_get_kv_type(gguf_ctx.get(), arch_id) != GGUF_TYPE_STRING) {
return COMMON_DECISION_TYPE_UNKNOWN; // malformed metadata
}
arch = gguf_get_val_str(gguf_ctx.get(), arch_id);
const std::string arch = gguf_get_val_str(gguf_ctx.get(), arch_id);
if (arch.empty()) {
return COMMON_DECISION_TYPE_UNKNOWN;
return info;
}
const std::string key = arch + ".decision.type";
const int64_t type_id = gguf_find_key(gguf_ctx.get(), key.c_str());
const int64_t type_id = gguf_find_key(gguf_ctx.get(), (arch + ".decision.type").c_str());
if (type_id < 0) {
return COMMON_DECISION_TYPE_NONE;
info.decision_type = COMMON_DECISION_TYPE_NONE;
} else if (gguf_get_kv_type(gguf_ctx.get(), type_id) == GGUF_TYPE_STRING) {
info.decision_type = common_decision_type_from_string(gguf_get_val_str(gguf_ctx.get(), type_id));
}
if (gguf_get_kv_type(gguf_ctx.get(), type_id) != GGUF_TYPE_STRING) {
return COMMON_DECISION_TYPE_UNKNOWN; // malformed metadata
// same key and type as the model loader
const int64_t ctx_id = gguf_find_key(gguf_ctx.get(), (arch + ".context_length").c_str());
if (ctx_id >= 0 && gguf_get_kv_type(gguf_ctx.get(), ctx_id) == GGUF_TYPE_UINT32) {
info.n_ctx_train = gguf_get_val_u32(gguf_ctx.get(), ctx_id);
}
return common_decision_type_from_string(gguf_get_val_str(gguf_ctx.get(), type_id));
return info;
}
common_init_result::common_init_result(common_params & params, bool model_only) :
+7 -3
View File
@@ -970,9 +970,13 @@ enum common_decision_type {
common_decision_type common_get_decision_type(const struct llama_model * model);
// same as above, but reads a GGUF file; it does not load the model
// returns COMMON_DECISION_TYPE_UNKNOWN if the file is missing, unreadable, or invalid
common_decision_type common_get_decision_type(const std::string & fname);
// metadata of a GGUF file, read without loading the model
struct common_gguf_info {
common_decision_type decision_type = COMMON_DECISION_TYPE_UNKNOWN; // UNKNOWN if the file is missing, unreadable, or invalid
uint32_t n_ctx_train = 0; // 0 if unknown
};
common_gguf_info common_get_gguf_info(const std::string & fname);
// note: defines the model, context, samplers, ets. lifetimes
struct common_init_result {
+1 -1
View File
@@ -107,7 +107,7 @@ static __device__ __forceinline__ float op_ceil(float x) {
}
static __device__ __forceinline__ float op_round(float x) {
return round(x);
return roundf(x);
}
static __device__ __forceinline__ float op_trunc(float x) {
+58 -25
View File
@@ -2471,19 +2471,57 @@ ggml_tensor * llm_graph_context::build_inp_embd(ggml_tensor * tok_embd, float to
auto inp = std::make_unique<llm_graph_input_embd>(n_embd_inp);
inp->tokens = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, ubatch.n_tokens);
cb(inp->tokens, "inp_tokens", -1);
ggml_set_input(inp->tokens);
res->t_inp_tokens = inp->tokens;
// mixed path (ubatch.is_mixed()): set_rows the token rows into a copy of the embd rows, with its own inputs as select branches must not share tensors
// TODO: use inp->tokens and inp->embd once ggml_build_forward_select allows it
const bool has_mixed = llm_arch_supports_mixed_batch(arch) && cparams.ctx_type == LLAMA_CONTEXT_TYPE_DEFAULT;
inp->embd = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, n_embd_inp, ubatch.n_tokens);
cb(inp->embd, "inp_embd", -1);
ggml_set_input(inp->embd);
const int64_t n_tok_rows = has_mixed ? llm_graph_n_tok_rows(ubatch) : 0;
// token embeddings with lora and padding
auto build_tok = [&](ggml_tensor * ids) {
ggml_tensor * cur = ggml_get_rows(ctx0, tok_embd, ids);
// we have 3 standard paths to produce the input embeddings for the first layer:
// - embd0: extract from the token embeddings weight (`tok_embd`) using the input token ids
// - embd1: pass raw embeddings, skipping the `tok_embd`
// - embd2: mixed path of both tokens ids + raw embeddings (if supported)
ggml_tensor * embd0 = nullptr;
ggml_tensor * embd1 = nullptr;
ggml_tensor * embd2 = nullptr;
// construct the input tensors
{
inp->tokens = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, ubatch.n_tokens);
cb(inp->tokens, "inp_tokens", -1);
ggml_set_input(inp->tokens);
res->t_inp_tokens = inp->tokens;
inp->embd = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, n_embd_inp, ubatch.n_tokens);
cb(inp->embd, "inp_embd", -1);
ggml_set_input(inp->embd);
if (has_mixed) {
inp->mixed_tokens = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, n_tok_rows);
cb(inp->mixed_tokens, "inp_mixed_tokens", -1);
ggml_set_input(inp->mixed_tokens);
}
}
// the embeddings placeholders for the 3 paths
// we use ggml_build_forward_order to make the GET_ROWS ops stick at the beginning of the compute graph
// this way the embeddings remain in the host buffer, and the GET_ROWS run before any other computations
{
embd0 = ggml_get_rows(ctx0, tok_embd, inp->tokens);
ggml_build_forward_order(gf, embd0);
embd1 = inp->embd;
if (has_mixed) {
embd2 = ggml_get_rows(ctx0, tok_embd, inp->mixed_tokens);
ggml_build_forward_order(gf, embd2);
}
}
// helper for extracting token embeddings with lora and padding
// TODO: when lora is active, this is likely going to cause issues similar to https://github.com/ggml-org/llama.cpp/pull/30160
// need to add lora tests and refactor the logic to make the lora GET_ROWS go at the front of the graph
auto build_tok = [&](ggml_tensor * cur, ggml_tensor * ids) {
// apply lora for embedding tokens if needed
for (const auto & lora : *loras) {
llama_adapter_lora_weight * lw = lora.first->get_weight(tok_embd);
@@ -2514,21 +2552,15 @@ ggml_tensor * llm_graph_context::build_inp_embd(ggml_tensor * tok_embd, float to
std::array<ggml_tensor *, 3> inps = {};
// token embeddings path (ubatch.token != nullptr)
inps[0] = build_tok(inp->tokens);
inps[0] = build_tok(embd0, inp->tokens);
// vector embeddings path (ubatch.embd != nullptr)
inps[1] = inp->embd;
inps[1] = embd1;
assert(ggml_are_same_shape (inps[0], inps[1]));
assert(ggml_are_same_stride(inps[0], inps[1]));
// mixed path (ubatch.is_mixed()): set_rows the token rows into a copy of the embd rows, with its own inputs as select branches must not share tensors
// TODO: use inp->tokens and inp->embd once ggml_build_forward_select allows it
const bool has_mixed = llm_arch_supports_mixed_batch(arch) && cparams.ctx_type == LLAMA_CONTEXT_TYPE_DEFAULT;
if (has_mixed) {
const int64_t n_tok_rows = llm_graph_n_tok_rows(ubatch);
inp->mixed_tokens = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, n_tok_rows);
cb(inp->mixed_tokens, "inp_mixed_tokens", -1);
ggml_set_input(inp->mixed_tokens);
inp->mixed_slots = ggml_new_tensor_1d(ctx0, GGML_TYPE_I64, n_tok_rows);
cb(inp->mixed_slots, "inp_mixed_slots", -1);
ggml_set_input(inp->mixed_slots);
@@ -2538,11 +2570,12 @@ ggml_tensor * llm_graph_context::build_inp_embd(ggml_tensor * tok_embd, float to
ggml_set_input(inp->mixed_embd);
// note: set_rows writes into its destination, so it gets a copy of the input
inps[2] = ggml_set_rows(ctx0, ggml_dup(ctx0, inp->mixed_embd), build_tok(inp->mixed_tokens), inp->mixed_slots);
}
ggml_tensor * embd_mixed = build_tok(embd2, inp->mixed_tokens);
inps[2] = ggml_set_rows(ctx0, ggml_dup(ctx0, inp->mixed_embd), embd_mixed, inp->mixed_slots);
assert(ggml_are_same_shape (inps[0], inps[1]));
assert(ggml_are_same_stride(inps[0], inps[1]));
assert(ggml_are_same_shape (inps[0], inps[2]));
assert(ggml_are_same_stride(inps[0], inps[2]));
}
const int idx = ubatch.is_mixed() ? 2 : ubatch.token ? 0 : 1;
+26 -18
View File
@@ -159,6 +159,14 @@ llama_model_gemma4::graph::graph(const llama_model & model, const llm_graph_para
ggml_tensor * cur;
ggml_tensor * inpL;
// do the PLE first to guarantee it is done in the host buffer
// ref: https://github.com/ggml-org/llama.cpp/pull/30160
ggml_tensor * inp_per_layer = nullptr;
if (model.per_layer_tok_embd) {
inp_per_layer = build_inp_per_layer();
ggml_build_forward_expand(gf, inp_per_layer);
}
// important: do not normalize weights for raw embeddings input (i.e. encoded image emdeddings)
inpL = build_inp_embd(model.tok_embd, sqrtf(n_embd));
cb(inpL, "inp_scaled", -1);
@@ -171,10 +179,11 @@ llama_model_gemma4::graph::graph(const llama_model & model, const llm_graph_para
ggml_tensor * inp_out_ids = build_inp_out_ids();
ggml_tensor * inp_per_layer = nullptr;
if (model.per_layer_tok_embd) {
inp_per_layer = build_inp_per_layer();
ggml_build_forward_expand(gf, inp_per_layer);
const float tok_embd_scale = sqrtf((float) n_embd_per_layer);
inp_per_layer = ggml_scale (ctx0, inp_per_layer, tok_embd_scale);
inp_per_layer = ggml_reshape_3d(ctx0, inp_per_layer, n_embd_per_layer, n_layer, inp_per_layer->ne[1]);
// inp_per_layer shape: [n_embd_per_layer, n_tokens, n_layer]
inp_per_layer = project_per_layer_inputs(inpL, inp_per_layer);
@@ -448,10 +457,13 @@ public:
llama_prefetch_rows(ple, ubatch->token, ubatch->n_tokens);
}
ggml_backend_tensor_set(tokens, ubatch->token, 0, ubatch->n_tokens * ggml_element_size(tokens));
} else if (prefetch) {
// [TAG_GEMMA4_IMG_PADDING]
} else {
const int32_t padding = 0;
llama_prefetch_rows(ple, &padding, 1);
if (prefetch) {
// [TAG_GEMMA4_IMG_PADDING]
llama_prefetch_rows(ple, &padding, 1);
}
ggml_backend_tensor_set(token0, &padding, 0, ggml_element_size(token0));
}
}
@@ -460,6 +472,7 @@ public:
}
ggml_tensor * tokens = nullptr;
ggml_tensor * token0 = nullptr;
const llama_model & model;
};
@@ -470,30 +483,25 @@ ggml_tensor * llama_model_gemma4::graph::build_inp_per_layer() {
auto inp = std::make_unique<llm_graph_input_gemma4_ple>(model);
ggml_tensor * inp_per_layer;
float tok_embd_scale = sqrtf((float) n_embd_per_layer);
// mixed ubatch: embd rows have token id 0, same padding row as below
// TODO: use ggml_build_forward_select
if (ubatch.token) {
inp->tokens = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, ubatch.n_tokens);
ggml_set_input(inp->tokens);
res->t_inp_tokens = inp->tokens;
inp_per_layer = ggml_get_rows (ctx0, model.per_layer_tok_embd, inp->tokens);
inp_per_layer = ggml_reshape_3d(ctx0, inp_per_layer, n_embd_per_layer, n_layer, n_tokens);
inp_per_layer = ggml_scale (ctx0, inp_per_layer, tok_embd_scale);
inp_per_layer = ggml_get_rows(ctx0, model.per_layer_tok_embd, inp->tokens);
cb(inp_per_layer, "inp_per_layer_selected", -1);
} else {
// [TAG_GEMMA4_IMG_PADDING]
// Multimodal embedding path: use padding token (ID=0) embedding
// TODO: verify if this is the correct behavior in transformers implementation
const int64_t embd_size = model.per_layer_tok_embd->ne[0]; // n_embd_per_layer * n_layer
// Extract and dequantize padding token embedding (row 0)
ggml_tensor * padding = ggml_view_1d(ctx0, model.per_layer_tok_embd, embd_size, 0);
inp_per_layer = ggml_cast (ctx0, padding, GGML_TYPE_F32);
inp_per_layer = ggml_scale(ctx0, inp_per_layer, tok_embd_scale);
// [TAG_GEMMA4_IMG_PADDING]
inp->token0 = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, 1);
ggml_set_input(inp->token0);
res->t_inp_tokens = inp->token0;
// Reshape to [n_embd_per_layer, n_layer, 1]
inp_per_layer = ggml_reshape_3d(ctx0, inp_per_layer, n_embd_per_layer, n_layer, 1);
inp_per_layer = ggml_get_rows(ctx0, model.per_layer_tok_embd, inp->token0);
cb(inp_per_layer, "inp_per_layer_multimodal", -1);
}
res->add_input(std::move(inp));
+11 -11
View File
@@ -405,10 +405,6 @@ llama_model_qwen4exp::graph::graph(const llama_model & model, const llm_graph_pa
int sections[4];
std::copy(std::begin(hparams.rope_sections), std::begin(hparams.rope_sections) + 4, sections);
ggml_tensor * inpL = build_inp_embd(model.tok_embd);
cb(inpL, "model.input_embed", -1);
ggml_build_forward_expand(gf, inpL);
auto * inp = build_inp_mem_hybrid();
// qwen4exp always builds llama_memory_hybrid_idx, so this downcast is safe
@@ -421,6 +417,17 @@ llama_model_qwen4exp::graph::graph(const llama_model & model, const llm_graph_pa
"the indexer cache must track the attention cache cell for cell");
}
ggml_tensor * ple_emb = nullptr;
if (hparams.ple_n_heads > 0) {
ple_emb = build_inp_ple(mctx_hyb);
// make sure ple_emb and build_inp_embd are in the same graph split
ggml_build_forward_expand(gf, ple_emb);
}
ggml_tensor * inpL = build_inp_embd(model.tok_embd);
cb(inpL, "model.input_embed", -1);
ggml_build_forward_expand(gf, inpL);
// the QSA layers share one set of k-pool inputs
// the CUDA lightning indexer takes 32 or 64 heads, QSA has a few, so it scores with plain ops
llm_graph_input_kpool * inp_kpool = nullptr;
@@ -431,13 +438,6 @@ llama_model_qwen4exp::graph::graph(const llama_model & model, const llm_graph_pa
ggml_tensor * inp_pos = build_inp_pos();
ggml_tensor * inp_out_ids = build_inp_out_ids();
ggml_tensor * ple_emb = nullptr;
if (hparams.ple_n_heads > 0) {
ple_emb = build_inp_ple(mctx_hyb);
// make sure ple_emb and build_inp_embd are in the same graph split
ggml_build_forward_expand(gf, ple_emb);
}
// the wide residual starts as hc identical copies of the embedding
ggml_tensor * res_hc = ggml_repeat_4d(ctx0,
ggml_reshape_3d(ctx0, inpL, n_embd, 1, n_tokens),
+10 -3
View File
@@ -540,6 +540,7 @@ void server_model_meta::update_args(common_preset_context & ctx_preset, std::str
void server_model_meta::update_caps(const common_params & base) {
// reset to the default so a failed refresh cannot keep old values
architecture = server_model_architecture_json(false, false, false, {"text"});
n_ctx_train = 0;
// resolve the model file offline; do not download
common_params params;
@@ -564,10 +565,12 @@ void server_model_meta::update_caps(const common_params & base) {
return;
}
// read the output modalities from the GGUF metadata
// read the output modalities and the trained context from the GGUF metadata
std::vector<std::string> output_modalities = {"text"};
if (!params.model.path.empty()) {
output_modalities = server_model_output_modalities(common_get_decision_type(params.model.path));
const common_gguf_info info = common_get_gguf_info(params.model.path);
output_modalities = server_model_output_modalities(info.decision_type);
n_ctx_train = info.n_ctx_train;
}
bool inp_image = false;
@@ -2122,9 +2125,13 @@ void server_models_routes::init_routes() {
{"source", server_model_source_to_string(meta.source)},
{"can_remove", meta.source == SERVER_MODEL_SOURCE_CACHE},
// {"need_download", meta.need_download},
// TODO: add other fields, may require reading GGUF metadata
// TODO: add other fields from the GGUF metadata
};
if (meta.n_ctx_train > 0) {
model_info["context_length"] = meta.n_ctx_train;
}
// merge with loaded_info from the child process if available
if (meta.is_running()) {
for (auto it = meta.loaded_info.begin(); it != meta.loaded_info.end(); ++it) {
+1
View File
@@ -86,6 +86,7 @@ struct server_model_meta {
int stop_timeout = 0; // seconds to wait before force-killing the model instance during shutdown
bool hidden = false; // hidden from GET /models, but still accept if requested
json architecture = server_model_architecture_json(false, false, false, {"text"});
uint32_t n_ctx_train = 0; // trained context, read from the GGUF metadata; 0 when unknown
bool is_ready() const {
return status == SERVER_MODEL_STATUS_LOADED;
@@ -12,37 +12,38 @@
}
let { isFav, option, revealOnHover = true }: Props = $props();
// the favorite heart stays visible, so the reveal applies to the other icons
const revealClass =
'pointer-events-none opacity-0 group-hover:pointer-events-auto group-hover:opacity-100 [@media(pointer:coarse)]:pointer-events-auto [@media(pointer:coarse)]:opacity-100';
</script>
<div
class={[
'flex items-center justify-center gap-1 max-md:gap-2.5',
revealOnHover
? 'pointer-events-none opacity-0 group-hover:pointer-events-auto group-hover:opacity-100 [@media(pointer:coarse)]:pointer-events-auto [@media(pointer:coarse)]:opacity-100'
: ''
]}
class="flex items-center justify-center gap-1 max-md:gap-2.5"
onclick={(event) => event.stopPropagation()}
onkeydown={(event) => event.stopPropagation()}
role="presentation"
>
<ActionIcon
class="h-5 w-5 hover:text-foreground"
icon={Info}
iconSize="h-4 w-4"
onclick={() =>
// a phone has no manager: its information dialog takes the icon
deviceStore.isMobile
? uiStore.openModelInformation(option)
: uiStore.openModelsManager(option.id)}
tooltip="Manage model"
tooltipAsTitle
/>
<span class={revealOnHover ? revealClass : ''}>
<ActionIcon
class="h-5 w-5 hover:text-foreground"
icon={Info}
iconSize="h-4 w-4"
onclick={() =>
// a phone has no manager: its information dialog takes the icon
deviceStore.isMobile
? uiStore.openModelInformation(option)
: uiStore.openModelsManager(option.id)}
tooltip="Manage model"
tooltipAsTitle
/>
</span>
{#if isFav}
<span class="flex h-5 w-5 items-center justify-center">
<span class="flex group-hover:hidden [@media(pointer:coarse)]:hidden">
<ActionIcon
class="h-5 w-5 text-rose-500 hover:text-foreground"
class="h-5 w-5 text-rose-500"
icon={Heart}
iconSize="h-4 w-4"
onclick={() => modelsStore.toggleFavorite(option.model)}
@@ -63,13 +64,15 @@
</span>
</span>
{:else}
<ActionIcon
class="h-5 w-5 hover:text-foreground"
icon={Heart}
iconSize="h-4 w-4"
onclick={() => modelsStore.toggleFavorite(option.model)}
tooltip="Add to favorites"
tooltipAsTitle
/>
<span class={revealOnHover ? revealClass : ''}>
<ActionIcon
class="h-5 w-5 hover:text-foreground"
icon={Heart}
iconSize="h-4 w-4"
onclick={() => modelsStore.toggleFavorite(option.model)}
tooltip="Add to favorites"
tooltipAsTitle
/>
</span>
{/if}
</div>
@@ -16,7 +16,7 @@
import type { ModelOption } from '$lib/types/models';
import { filterModelOptions } from '$lib/utils';
import { type Snippet, untrack } from 'svelte';
import { SvelteMap, SvelteSet } from 'svelte/reactivity';
import { SvelteMap } from 'svelte/reactivity';
interface Props {
class?: string;
@@ -170,16 +170,29 @@
if (remaining.length > 0) rest.push({ ...entry, base: remaining[0], quants: remaining });
}
const claimed = new SvelteSet<string>();
const favorites = rest.filter((entry) =>
entry.quants.some((q) => modelsStore.favoriteModelIds.has(q.model))
);
// a favorite quant stands on its own, listed flat like a loaded one, and the
// quants left behind stay with their repo in the local block
const favorites: ModelQuantGroup[] = [];
const localRest: ModelQuantGroup[] = [];
for (const entry of favorites) claimed.add(entry.key);
for (const entry of rest) {
for (const quant of entry.quants) {
if (modelsStore.favoriteModelIds.has(quant.model)) {
favorites.push({ ...entry, base: quant, key: quant.id, quants: [quant] });
}
}
const { hidden, local } = splitHiddenQuants(
rest.filter((entry) => !claimed.has(entry.key)),
(option) => modelsStore.isHidden(option.id)
const remaining = entry.quants.filter(
(quant) => !modelsStore.favoriteModelIds.has(quant.model)
);
if (remaining.length > 0) {
localRest.push({ ...entry, base: remaining[0], quants: remaining });
}
}
const { hidden, local } = splitHiddenQuants(localRest, (option) =>
modelsStore.isHidden(option.id)
);
const ordered: ModelsTableGroup[] = [];
// loaded models lead the table, then favorites, then the local block
@@ -7,7 +7,7 @@
import ModelsManagerStatusCell from './ModelsManagerStatusCell.svelte';
import { modelRowActions } from './row-actions';
import { configuredContext, downloadProgressFor } from './utils';
import { MoreHorizontal } from '@lucide/svelte';
import { Heart, MoreHorizontal } from '@lucide/svelte';
import { DropdownMenuActions } from '$lib/components/app';
import { MODEL_ROW_GRID_CLASS, MODEL_ROW_TRAILING_CELL_CLASS } from '$lib/constants';
import { ModelRowDownloadState } from '$lib/enums';
@@ -74,6 +74,13 @@
<!-- a phone has no width for the modality icons, the id needs it more -->
<ModelCapabilities hideModalities={deviceStore.isMobile} {option} />
{#if favorite}
<!-- the heart is decorative, the button's text carries the state -->
<Heart aria-hidden="true" class="h-3.5 w-3.5 shrink-0 text-rose-500" />
<span class="sr-only">favorited</span>
{/if}
</span>
</button>
@@ -7,8 +7,7 @@
hasActiveFilters,
modelContextLength,
type ModelQuantGroup,
type ModelsTableGroup,
statusRank
type ModelsTableGroup
} from './utils';
import {
ArrowDown,
@@ -155,10 +154,6 @@
return (modelContextLength(a) ?? 0) - (modelContextLength(b) ?? 0);
case ModelsTableSortKey.NAME:
return a.model.localeCompare(b.model);
case ModelsTableSortKey.STATUS:
// a running model leads, then one that is being worked on (loading,
// sleeping), then the rest; the reported status sorts the row's own cell
return statusRank(b) - statusRank(a);
default:
return 0;
}
@@ -357,9 +352,7 @@
{@render sortHeader(ModelsTableSortKey.CONTEXT, 'Context')}
</span>
<span class="justify-self-center max-md:hidden">
{@render sortHeader(ModelsTableSortKey.STATUS, 'Status')}
</span>
<span class="justify-self-center max-md:hidden">Status</span>
<span class="text-center max-md:hidden">Actions</span>
</div>
@@ -1,10 +1,5 @@
import { LOCAL_BACKEND_ID, type ModalityKey } from '$lib/constants';
import {
ModelCapability,
ModelGroupKind,
ModelsTableGroupKind,
ServerModelStatus
} from '$lib/enums';
import { ModelCapability, ModelGroupKind, ModelsTableGroupKind } from '$lib/enums';
import { HuggingFaceService, ModelsService } from '$lib/services';
import { modelsStore } from '$lib/stores';
import type { ModelDownloadEntry, ModelDownloadProgress, ModelOption } from '$lib/types/models';
@@ -27,20 +22,6 @@ export function hasActiveFilters(
return contextLimit > 0 || modalities.length > 0 || capabilities.length > 0;
}
/**
* Order a status sorts behind: loaded first, then a model being worked on
* (loading, sleeping), then the rest.
*/
export function statusRank(option: ModelOption): number {
const status = modelsStore.getModelStatus(option.model);
if (modelsStore.isModelRunning(option.model)) return 2;
if (status === ServerModelStatus.LOADING || status === ServerModelStatus.SLEEPING) return 1;
return 0;
}
/** Byte counts of a tracked download: live while it runs, frozen while paused. */
export function downloadProgressFor(repoWithTag: string): ModelDownloadProgress | null {
return (
@@ -177,7 +177,7 @@
<DropdownMenu.Content
align="end"
class="w-full md:min-w-80 md:w-112 max-w-[calc(100vw-2rem)] p-0! max-h-[min(40rem,calc(var(--bits-dropdown-menu-content-available-height)-1rem))]"
class="w-full md:min-w-80 md:w-md max-w-[calc(100vw-2rem)] p-0! max-h-[min(40rem,calc(var(--bits-dropdown-menu-content-available-height)-1rem))]"
onOpenAutoFocus={(event) => event.preventDefault()}
>
<DropdownMenuSearchable
@@ -17,7 +17,7 @@
<DropdownMenuPrimitive.Content
bind:ref
class={cn(
'z-50 max-h-[calc(var(--bits-dropdown-menu-content-available-height)-1rem)] min-w-[8rem] origin-(--bits-dropdown-menu-content-transform-origin) overflow-x-hidden overflow-y-auto rounded-md border border-border bg-popover p-1.5 text-popover-foreground shadow-md outline-none data-[side=bottom]:slide-in-from-top-2 data-[side=left]:slide-in-from-right-2 data-[side=right]:slide-in-from-left-2 data-[side=top]:slide-in-from-bottom-2 data-[state=closed]:animate-out data-[state=closed]:fade-out-0 data-[state=closed]:fill-mode-forwards data-[state=closed]:zoom-out-95 data-[state=open]:animate-in data-[state=open]:fade-in-0 data-[state=open]:zoom-in-95 dark:border-border/20',
'z-50 max-h-[calc(var(--bits-dropdown-menu-content-available-height)-1rem)] min-w-[8rem] origin-(--bits-dropdown-menu-content-transform-origin) overflow-x-hidden overflow-y-auto rounded-xl border border-border bg-popover p-1.5 text-popover-foreground shadow-md outline-none data-[side=bottom]:slide-in-from-top-2 data-[side=left]:slide-in-from-right-2 data-[side=right]:slide-in-from-left-2 data-[side=top]:slide-in-from-bottom-2 data-[state=closed]:animate-out data-[state=closed]:fade-out-0 data-[state=closed]:fill-mode-forwards data-[state=closed]:zoom-out-95 data-[state=open]:animate-in data-[state=open]:fade-in-0 data-[state=open]:zoom-in-95 dark:border-border/20',
className
)}
data-slot="dropdown-menu-content"
@@ -17,7 +17,7 @@
<DropdownMenuPrimitive.Item
bind:ref
class={cn(
"relative flex cursor-pointer items-center gap-2 rounded-sm px-2 py-1.5 text-sm outline-hidden select-none data-highlighted:bg-accent data-highlighted:text-accent-foreground data-[disabled]:pointer-events-none data-[disabled]:opacity-50 data-[inset]:pl-8 data-[variant=destructive]:text-destructive data-[variant=destructive]:data-highlighted:bg-destructive/10 data-[variant=destructive]:data-highlighted:text-destructive dark:data-[variant=destructive]:data-highlighted:bg-destructive/20 [&_svg]:pointer-events-none [&_svg]:shrink-0 [&_svg:not([class*='size-'])]:size-4 [&_svg:not([class*='text-'])]:text-muted-foreground data-[variant=destructive]:*:[svg]:!text-destructive",
"relative flex cursor-pointer items-center gap-2 rounded-md px-2 py-1.5 text-sm outline-hidden select-none data-highlighted:bg-accent data-highlighted:text-accent-foreground data-[disabled]:pointer-events-none data-[disabled]:opacity-50 data-[inset]:pl-8 data-[variant=destructive]:text-destructive data-[variant=destructive]:data-highlighted:bg-destructive/10 data-[variant=destructive]:data-highlighted:text-destructive dark:data-[variant=destructive]:data-highlighted:bg-destructive/20 [&_svg]:pointer-events-none [&_svg]:shrink-0 [&_svg:not([class*='size-'])]:size-4 [&_svg:not([class*='text-'])]:text-muted-foreground data-[variant=destructive]:*:[svg]:!text-destructive",
className
)}
data-inset={inset}
@@ -129,7 +129,7 @@ export const SETTINGS_REGISTRY: SettingsSectionEntry[] = [
// Deliberately off for now: the natural place to turn it on is the first
// run experience, once onboarding exists to ask the user about it.
defaultValue: false,
help: 'Fetch model metadata (avatars, context length, chat template, file sizes) from the Hugging Face Hub. When off, the UI only shows what the server reports for /v1/models and hides the org avatars.',
help: 'Fetch model metadata (avatars, context length, chat template) from the Hugging Face Hub. When off, the UI only shows what the server reports for /v1/models and hides the org avatars.',
key: SETTINGS_KEYS.USE_HUGGING_FACE_HUB,
label: 'Use Hugging Face Hub API for models metadata',
type: SettingsFieldType.CHECKBOX
+1 -2
View File
@@ -80,6 +80,5 @@ export enum ModelRowDownloadState {
/** Column the models manager table can be ordered by. */
export enum ModelsTableSortKey {
CONTEXT = 'context',
NAME = 'name',
STATUS = 'status'
NAME = 'name'
}
@@ -622,6 +622,13 @@ class ModelsStore implements ModelPropsHost, ModelStatusHost {
capabilities: rawCapabilities.filter((value: unknown): value is string =>
Boolean(value)
),
// 0 is the server's way of leaving the trained context unknown; a
// model-mode listing reports it as meta.n_ctx_train instead
contextLength:
item.context_length ||
(typeof item.meta?.n_ctx_train === 'number' && item.meta.n_ctx_train > 0
? item.meta.n_ctx_train
: undefined),
description: details?.description,
details: details?.details,
draftSidecars: mergedDraftSidecars(
+2
View File
@@ -100,6 +100,8 @@ export interface ApiModelDataEntry {
tags?: string[];
/** Modality capabilities, reported by the router for every model regardless of load state */
architecture?: ApiModelArchitecture;
/** Trained context of the model, read from its GGUF metadata at registration */
context_length?: number;
/** Legacy meta field (may be present in older responses) */
meta?: Record<string, unknown> | null;
}
@@ -0,0 +1,243 @@
// Guards the manager table ordering and filtering: the context column sorts and
// filters once a model's context is known, from the option or the Hub record.
import ModelsManagerWrapper from './components/ModelsManagerWrapper.svelte';
import { SETTINGS_KEYS } from '$lib/constants';
import { ServerModelStatus } from '$lib/enums';
import { HuggingFaceService } from '$lib/services';
import { modelsStore, settingsStore } from '$lib/stores';
import type { ApiModelDataEntry } from '$lib/types';
import type { ModelOption } from '$lib/types/models';
import { SvelteMap } from 'svelte/reactivity';
import { beforeEach, expect, it, vi } from 'vitest';
import { render } from 'vitest-browser-svelte';
function option(model: string, contextLength?: number): ModelOption {
return {
capabilities: [],
contextLength,
id: model,
model,
name: model
};
}
const models = [
option('org/alpha-8b:Q4_K_M', 8192),
option('org/beta-8b:Q4_K_M', 131072),
option('org/gamma-8b:Q4_K_M', 32768)
];
// the router listing carries no context, so a row starts without one
const modelsWithoutContext = models.map((model) => ({ ...model, contextLength: undefined }));
beforeEach(() => {
modelsStore.routerModels = [];
modelsStore.models = modelsWithoutContext;
modelsStore.favoriteModelIds = new Set();
settingsStore.config[SETTINGS_KEYS.GROUP_MODELS_BY_FAMILY] = false;
});
/** Renders the manager and waits for the test models to show. */
async function renderWithModels(rows: ModelOption[] = modelsWithoutContext) {
const screen = render(ModelsManagerWrapper);
modelsStore.models = rows;
await expect.element(screen.getByText(/gamma\s+8B/)).toBeVisible();
return screen;
}
function rowNames(container: HTMLElement): string[] {
return [...container.querySelectorAll('button[aria-pressed]')].map(
(row) => row.textContent ?? ''
);
}
/** The Hub details cache the rows read their context from. */
function detailsCache() {
return (
HuggingFaceService as unknown as {
detailsCache: SvelteMap<string, { gguf?: { context_length?: number } } | null>;
}
).detailsCache;
}
function warmCache() {
const cache = detailsCache();
cache.set('org/alpha-8b', { gguf: { context_length: 8192 } });
cache.set('org/beta-8b', { gguf: { context_length: 131072 } });
cache.set('org/gamma-8b', { gguf: { context_length: 32768 } });
}
it('sorts by context', async () => {
const screen = await renderWithModels(models);
// lowest first
await screen.getByTitle('Sort by context, lowest first').click();
const ascending = rowNames(screen.container);
await screen.getByTitle('Sort by context, highest first').click();
const descending = rowNames(screen.container);
expect(ascending).not.toEqual(descending);
});
it('sorts by context with family grouping on', async () => {
settingsStore.config[SETTINGS_KEYS.GROUP_MODELS_BY_FAMILY] = true;
const screen = await renderWithModels(models);
// lowest first
await screen.getByTitle('Sort by context, lowest first').click();
const ascending = rowNames(screen.container);
await screen.getByTitle('Sort by context, highest first').click();
const descending = rowNames(screen.container);
expect(ascending).not.toEqual(descending);
});
it('lists the favorited quants of a repo as flat rows', async () => {
// one quant of the beta repo is a favorite, the other stays with the repo
modelsStore.favoriteModelIds = new Set(['org/alpha-8b:Q4_K_M', 'org/beta-8b:Q4_K_M']);
const screen = await renderWithModels([
modelsWithoutContext[0],
modelsWithoutContext[1],
option('org/beta-8b:Q8_0'),
modelsWithoutContext[2]
]);
// the favorite quant is a model row of its own, not a repo with subitems
expect(screen.container.textContent).not.toContain('2 quants available');
// the repo appears once in favorites for its favorited quant and once in the
// local block for the quant left behind
expect(screen.getByText(/beta\s+8B/).elements().length).toBe(2);
});
/** A router listing entry carrying only the meta context. */
function entry(model: string, nCtxTrain: number): ApiModelDataEntry {
return {
created: 0,
id: model,
in_cache: false,
meta: { n_ctx_train: nCtxTrain },
object: 'model',
owned_by: 'llamacpp',
path: `/models/${model}`,
status: { value: ServerModelStatus.UNLOADED }
};
}
it('sorts by the meta context of a listing without the router field', async () => {
// a listing that skips the router's GGUF read reports the trained context
// only as meta.n_ctx_train, so the option mapping falls back to it
vi.spyOn(globalThis, 'fetch').mockImplementation(async (input: RequestInfo | URL) => {
const url = typeof input === 'string' ? input : input instanceof URL ? input.href : input.url;
if (url.includes('/props')) {
return new Response(
JSON.stringify({
default_generation_settings: { n_ctx: 0, params: {} },
model_alias: 'llama-server',
model_path: 'none',
role: 'router'
}),
{ headers: { 'Content-Type': 'application/json' }, status: 200 }
);
}
if (url.includes('/server')) {
return new Response(
JSON.stringify({ git_branch: 'test', git_commit: 'test', mode: 'router', version: 'test' }),
{ headers: { 'Content-Type': 'application/json' }, status: 200 }
);
}
if (/\/v1\/models|\/models\b/.test(url)) {
return new Response(
JSON.stringify({
data: [
entry('org/alpha-8b:Q4_K_M', 8192),
entry('org/beta-8b:Q4_K_M', 131072),
entry('org/gamma-8b:Q4_K_M', 32768)
],
object: 'list'
}),
{ headers: { 'Content-Type': 'application/json' }, status: 200 }
);
}
throw new Error(`unexpected fetch in the test: ${url}`);
});
const screen = render(ModelsManagerWrapper);
// the fetch maps the listing into options, the meta context fills in
await modelsStore.fetch(true);
await expect.element(screen.getByText(/gamma\s+8B/)).toBeVisible();
await screen.getByTitle('Sort by context, lowest first').click();
const names = rowNames(screen.container).join(' | ');
expect(names.indexOf('alpha')).toBeLessThan(names.indexOf('gamma'));
expect(names.indexOf('gamma')).toBeLessThan(names.indexOf('beta'));
});
it('re-sorts when the Hub details arrive after the sort was clicked', async () => {
const screen = await renderWithModels();
// the user sorts while the contexts are still unknown
await screen.getByTitle('Sort by context, lowest first').click();
// then the rows fetch their Hub records
warmCache();
// the table re-sorts once the cache answers
await vi.waitFor(() => {
const names = rowNames(screen.container).join(' | ');
expect(names.indexOf('alpha')).toBeLessThan(names.indexOf('gamma'));
expect(names.indexOf('gamma')).toBeLessThan(names.indexOf('beta'));
return names;
});
});
it('filters by search', async () => {
const screen = await renderWithModels();
await screen.getByPlaceholder('Search your models').fill('beta');
await expect.element(screen.getByText(/alpha\s+8B/)).not.toBeVisible();
await expect.element(screen.getByText(/beta\s+8B/)).toBeVisible();
});
it('filters by context and sorts from the Hub details cache', async () => {
warmCache();
const screen = await renderWithModels();
// open the context filter and ask for 32K or more
await screen.getByText('Context:').click();
await screen.getByText('32K or more').click();
await expect.element(screen.getByText(/alpha\s+8B/)).not.toBeVisible();
await expect.element(screen.getByText(/gamma\s+8B/)).toBeVisible();
// sorting re-runs once the cache answers
await screen.getByTitle('Sort by context, lowest first').click();
const names = rowNames(screen.container).join(' | ');
expect(names.indexOf('gamma')).toBeLessThan(names.indexOf('beta'));
});