From 803d02937cdd30791cffde6e24b59f77eba088c7 Mon Sep 17 00:00:00 2001 From: Ettore Di Giacinto Date: Wed, 17 Jun 2026 07:38:14 +0000 Subject: [PATCH] feat(ds4): wire SSD streaming + quality engine options, add 128GB DeepSeek gallery models The ds4 backend zero-initialized ds4_engine_options and exposed none of the engine's tunable knobs, so SSD streaming (run a model larger than RAM by streaming routed MoE experts from the GGUF on SSD) and the quality/perf knobs were unreachable from LocalAI model YAMLs. Map ModelOptions.Options onto ds4_engine_options through a declarative table (kEngineOptSpecs + apply_engine_option) instead of per-field branches: the struct is fixed C with no reflection, so the field set is enumerated once and a future knob is a one-line table row. Two fields use ds4's own typed parsers (GiB budgets, cache-experts count-or-NGB). Bare flags (e.g. "ssd_streaming") mean true; path-type options (mtp_path, expert_profile_path, directional_steering_file) resolve relative to the model directory so a gallery entry can reference a companion file by bare filename. mtp_draft/mtp_margin are now validated rather than parsed with throwing std::stoi/std::stof. Add gallery entries for the 128 GB class: - deepseek-v4-flash-q2-q4 (~91 GB, mixed q2/q4, fits RAM, higher quality) - deepseek-v4-flash-q4-ssd (~153 GB full 4-bit, runs on 128 GB via SSD streaming) - deepseek-v4-flash-q2-mtp (~81 GB + MTP speculative draft weights) - deepseek-v4-pro-q2-ssd (~433 GB Pro, experimental SSD streaming) SSD streaming is Metal (Darwin) only; the options are inert on CUDA/CPU. Signed-off-by: Ettore Di Giacinto Assisted-by: Claude:claude-opus-4-8 [Claude Code] --- .agents/ds4-backend.md | 33 ++++++ backend/cpp/ds4/grpc-server.cpp | 197 +++++++++++++++++++++++++++----- gallery/index.yaml | 106 +++++++++++++++++ 3 files changed, 306 insertions(+), 30 deletions(-) diff --git a/.agents/ds4-backend.md b/.agents/ds4-backend.md index 3edc14b11dbf..c1b649857c85 100644 --- a/.agents/ds4-backend.md +++ b/.agents/ds4-backend.md @@ -44,6 +44,39 @@ maps to `DS4_THINK_HIGH`. We pass the chosen mode to `ds4_chat_append_assistant_ via `ModelOptions.Options[] = "kv_cache_dir:/some/path"`. Format is **our own** - NOT bit-compatible with ds4-server's KVC files (interop is a follow-up plan). +## Engine options (LoadModel) + +`LoadModel` maps `ModelOptions.Options[]` (`"key:value"`, from model-YAML +`options:`) onto `ds4_engine_options` through a **declarative table** +(`kEngineOptSpecs` + `apply_engine_option` in `grpc-server.cpp`). The struct is +plain C with no reflection, so the field set is enumerated once in the table; +adding a future engine knob is a one-line table row, not a new branch. Unknown +keys are ignored (back-compat). A bare flag (`ssd_streaming` with no value) +means `true`. Path-type values (`mtp_path`, `expert_profile_path`, +`directional_steering_file`) resolve **relative to the model directory**, so a +gallery entry can reference a companion file it downloaded by bare filename; +absolute values pass through. `ds4_role` / `ds4_layers` / `ds4_listen` / +`ds4_route_timeout` / `kv_cache_dir` keep their dedicated handling (validation ++ coordinator wiring) and are not in the table. + +Wired keys: `mtp_path`, `mtp_draft`, `mtp_margin`, `prefill_chunk`, +`power_percent`, `warm_weights`, `quality`, `ssd_streaming`, +`ssd_streaming_cold`, `ssd_streaming_preload_experts`, +`ssd_streaming_cache_experts` (count or `NGB`, sets both experts+bytes via +`ds4_parse_streaming_cache_experts_arg`), `simulate_used_memory` (`NGB` via +`ds4_parse_gib_arg`), `expert_profile_path`, `directional_steering_file`, +`directional_steering_attn`, `directional_steering_ffn`. + +## SSD streaming (running models larger than RAM) + +ds4's **SSD streaming** keeps non-routed weights resident and streams routed MoE +experts from the GGUF on cache misses, turning "does it fit in RAM" into a speed +spectrum. **Metal (Darwin) only** - it is a no-op on CUDA/CPU. Enable with +`options: ["ssd_streaming"]`; size the routed-expert cache with +`ssd_streaming_cache_experts:NGB` (omit for ds4's automatic 80%-of-working-set +budget). Gallery entries built on this: `deepseek-v4-flash-q4-ssd` (153 GB Flash +on a 128 GB Mac) and `deepseek-v4-pro-q2-ssd` (433 GB Pro, experimental). + ## Build matrix | Build | Where | Notes | diff --git a/backend/cpp/ds4/grpc-server.cpp b/backend/cpp/ds4/grpc-server.cpp index f6672a4a0795..3d324cd6a7cb 100644 --- a/backend/cpp/ds4/grpc-server.cpp +++ b/backend/cpp/ds4/grpc-server.cpp @@ -25,6 +25,8 @@ extern "C" { #include #include #include +#include +#include #include #include #include @@ -105,6 +107,130 @@ static bool parse_layers_spec(const std::string &spec, ds4_distributed_layers *o return true; } +// Parse a boolean LoadModel option. An empty value (a bare flag-style option +// like "ssd_streaming" with no colon) means true so model YAMLs can write +// options: ["ssd_streaming"] to enable a switch. +static bool parse_bool_option(const std::string &s, bool *out) { + if (s.empty() || s == "true" || s == "1" || s == "yes" || s == "on") { *out = true; return true; } + if (s == "false" || s == "0" || s == "no" || s == "off") { *out = false; return true; } + return false; +} + +// Table-driven mapping from LoadModel option keys to ds4_engine_options fields. +// ds4_engine_options is a fixed C struct with no reflection, so the field set +// is enumerated once here; adding a future engine knob is a one-line table +// entry rather than a new branch in LoadModel. Two fields need ds4's own typed +// parsers (Gib, CacheExperts) so a plain string passthrough can't cover them. +enum class DsOptType { Bool, Int, Uint, Float, Str, Gib, CacheExperts }; + +struct DsOptSpec { + const char *key; + DsOptType type; + size_t off; // byte offset into ds4_engine_options + size_t off2; // second offset (CacheExperts writes experts + bytes) + bool is_path; // Str values: resolve a relative value against the model dir +}; + +static const DsOptSpec kEngineOptSpecs[] = { + {"mtp_path", DsOptType::Str, offsetof(ds4_engine_options, mtp_path), 0, true}, + {"mtp_draft", DsOptType::Int, offsetof(ds4_engine_options, mtp_draft_tokens), 0}, + {"mtp_margin", DsOptType::Float, offsetof(ds4_engine_options, mtp_margin), 0}, + {"prefill_chunk", DsOptType::Uint, offsetof(ds4_engine_options, prefill_chunk), 0}, + {"power_percent", DsOptType::Int, offsetof(ds4_engine_options, power_percent), 0}, + {"warm_weights", DsOptType::Bool, offsetof(ds4_engine_options, warm_weights), 0}, + {"quality", DsOptType::Bool, offsetof(ds4_engine_options, quality), 0}, + {"ssd_streaming", DsOptType::Bool, offsetof(ds4_engine_options, ssd_streaming), 0}, + {"ssd_streaming_cold", DsOptType::Bool, offsetof(ds4_engine_options, ssd_streaming_cold), 0}, + {"ssd_streaming_preload_experts", DsOptType::Uint, offsetof(ds4_engine_options, ssd_streaming_preload_experts), 0}, + {"ssd_streaming_cache_experts", DsOptType::CacheExperts, offsetof(ds4_engine_options, ssd_streaming_cache_experts), + offsetof(ds4_engine_options, ssd_streaming_cache_bytes)}, + {"simulate_used_memory", DsOptType::Gib, offsetof(ds4_engine_options, simulate_used_memory_bytes), 0}, + {"expert_profile_path", DsOptType::Str, offsetof(ds4_engine_options, expert_profile_path), 0, true}, + {"directional_steering_file", DsOptType::Str, offsetof(ds4_engine_options, directional_steering_file), 0, true}, + {"directional_steering_attn", DsOptType::Float, offsetof(ds4_engine_options, directional_steering_attn), 0}, + {"directional_steering_ffn", DsOptType::Float, offsetof(ds4_engine_options, directional_steering_ffn), 0}, +}; + +// Apply a single key:value LoadModel option to the engine options struct. +// Unknown keys are ignored (back-compat: callers pass mixed option sets). +// String values are copied into `storage`, whose elements the engine reads by +// pointer during ds4_engine_open; `storage` MUST have reserved capacity so +// push_back never reallocates and dangles an earlier c_str(). Returns false +// with `err` set when a recognized key has an invalid value. +static bool apply_engine_option(ds4_engine_options *opt, const std::string &key, + const std::string &val, const std::string &model_dir, + std::vector &storage, std::string &err) { + const DsOptSpec *spec = nullptr; + for (const auto &s : kEngineOptSpecs) { + if (key == s.key) { spec = &s; break; } + } + if (!spec) return true; // unknown key: ignore + + char *base = reinterpret_cast(opt); + switch (spec->type) { + case DsOptType::Bool: { + bool b = false; + if (!parse_bool_option(val, &b)) { err = key + " must be true/false"; return false; } + *reinterpret_cast(base + spec->off) = b; + return true; + } + case DsOptType::Int: { + char *end = nullptr; + long v = std::strtol(val.c_str(), &end, 10); + if (val.empty() || !end || *end != '\0') { err = key + " must be an integer"; return false; } + *reinterpret_cast(base + spec->off) = static_cast(v); + return true; + } + case DsOptType::Uint: { + char *end = nullptr; + long v = std::strtol(val.c_str(), &end, 10); + if (val.empty() || !end || *end != '\0' || v < 0 || v > static_cast(UINT32_MAX)) { + err = key + " must be a non-negative integer"; return false; + } + *reinterpret_cast(base + spec->off) = static_cast(v); + return true; + } + case DsOptType::Float: { + char *end = nullptr; + float f = std::strtof(val.c_str(), &end); + if (val.empty() || !end || *end != '\0') { err = key + " must be a number"; return false; } + *reinterpret_cast(base + spec->off) = f; + return true; + } + case DsOptType::Str: { + // Resolve a relative path option (e.g. mtp_path: a sibling GGUF the + // gallery downloaded next to the model) against the model directory, so + // YAMLs reference companion files by name. Absolute values pass through. + if (spec->is_path && !model_dir.empty() && !val.empty() && val.front() != '/') { + storage.push_back(model_dir + "/" + val); + } else { + storage.push_back(val); + } + *reinterpret_cast(base + spec->off) = storage.back().c_str(); + return true; + } + case DsOptType::Gib: { + uint64_t bytes = 0; + if (!ds4_parse_gib_arg(val.c_str(), &bytes)) { + err = key + " must be a GiB value, e.g. 64GB"; return false; + } + *reinterpret_cast(base + spec->off) = bytes; + return true; + } + case DsOptType::CacheExperts: { + uint32_t experts = 0; + uint64_t bytes = 0; + if (!ds4_parse_streaming_cache_experts_arg(val.c_str(), &experts, &bytes)) { + err = key + " must be a positive expert count or a GB budget"; return false; + } + *reinterpret_cast(base + spec->off) = experts; + *reinterpret_cast(base + spec->off2) = bytes; + return true; + } + } + return true; +} + // When acting as a distributed coordinator, block until the worker route // covers all layers (ds4_session_distributed_route_ready == 1) or the timeout // elapses. Returns an empty string on success, or an error message to return @@ -476,39 +602,10 @@ class DS4Backend final : public backend::Backend::Service { return GStatus::OK; } - std::string mtp_path; - int mtp_draft = 0; - float mtp_margin = 3.0f; - std::string ds4_role, ds4_layers, ds4_listen; - for (const auto &opt : request->options()) { - auto [k, v] = split_option(opt); - if (k == "mtp_path") mtp_path = v; - else if (k == "mtp_draft") mtp_draft = std::stoi(v); - else if (k == "mtp_margin") mtp_margin = std::stof(v); - else if (k == "kv_cache_dir") g_kv_cache_dir = v; - else if (k == "ds4_role") ds4_role = v; - else if (k == "ds4_layers") ds4_layers = v; - else if (k == "ds4_listen") ds4_listen = v; - else if (k == "ds4_route_timeout") { - if (!parse_positive_int(v, &g_route_timeout_sec)) { - result->set_success(false); - result->set_message("ds4: ds4_route_timeout must be a positive integer"); - return GStatus::OK; - } - } - } - - g_kv_cache.SetDir(g_kv_cache_dir); - ds4_engine_options opt = {}; opt.model_path = model_path.c_str(); - opt.mtp_path = mtp_path.empty() ? nullptr : mtp_path.c_str(); opt.n_threads = request->threads() > 0 ? request->threads() : 0; - opt.mtp_draft_tokens = mtp_draft; - opt.mtp_margin = mtp_margin; - opt.directional_steering_file = nullptr; - opt.warm_weights = false; - opt.quality = false; + opt.mtp_margin = 3.0f; // ds4 default; overridable via the mtp_margin option #if defined(DS4_NO_GPU) opt.backend = DS4_BACKEND_CPU; @@ -518,6 +615,46 @@ class DS4Backend final : public backend::Backend::Service { opt.backend = DS4_BACKEND_CUDA; #endif + // Stable storage for string-valued engine options. The engine reads + // these by pointer during ds4_engine_open, so the std::string backing + // store must outlive the call and not reallocate; reserve up front so + // push_back keeps every prior c_str() valid. Static + clear() reuses + // the buffer across LoadModel calls (the old engine is closed above). + static std::vector s_opt_strings; + s_opt_strings.clear(); + s_opt_strings.reserve(sizeof(kEngineOptSpecs) / sizeof(kEngineOptSpecs[0])); + + // Directory of the main model, used to resolve relative path options. + std::string model_dir; + if (auto slash = model_path.find_last_of('/'); slash != std::string::npos) { + model_dir = model_path.substr(0, slash); + } + + std::string ds4_role, ds4_layers, ds4_listen; + for (const auto &o : request->options()) { + auto [k, v] = split_option(o); + if (k == "kv_cache_dir") { g_kv_cache_dir = v; continue; } + else if (k == "ds4_role") { ds4_role = v; continue; } + else if (k == "ds4_layers") { ds4_layers = v; continue; } + else if (k == "ds4_listen") { ds4_listen = v; continue; } + else if (k == "ds4_route_timeout") { + if (!parse_positive_int(v, &g_route_timeout_sec)) { + result->set_success(false); + result->set_message("ds4: ds4_route_timeout must be a positive integer"); + return GStatus::OK; + } + continue; + } + std::string err; + if (!apply_engine_option(&opt, k, v, model_dir, s_opt_strings, err)) { + result->set_success(false); + result->set_message("ds4: " + err); + return GStatus::OK; + } + } + + g_kv_cache.SetDir(g_kv_cache_dir); + // Coordinator wiring. 'ds4_role:coordinator' enables layer-split // distributed inference: this process listens on ds4_listen and owns // the ds4_layers slice; workers dial in (see `local-ai worker diff --git a/gallery/index.yaml b/gallery/index.yaml index a82e025aa6b5..d9f6282f304d 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -33642,6 +33642,112 @@ - filename: ds4flash.gguf sha256: 31598c67c8b8744d3bcebcd19aa62253c6dc43cef3b8adf9f593656c9e86fd8c uri: huggingface://antirez/deepseek-v4-gguf/DeepSeek-V4-Flash-IQ2XXS-w2Q2K-AProjQ8-SExpQ8-OutQ8-chat-v2.gguf +- name: deepseek-v4-flash-q2-q4 + description: | + DeepSeek V4 Flash (mixed q2/q4 GGUF, ~91 GB) - only loadable via the ds4 backend. + The last 6 expert layers are kept at Q4_K (the rest IQ2XXS), trading a little + extra memory for higher quality than the pure-q2 build while still fitting in + RAM on a 128 GB machine. imatrix-tuned. Metal (Darwin) or CUDA (Linux). + See https://github.com/antirez/ds4 for details. + urls: + - https://huggingface.co/antirez/deepseek-v4-gguf + tags: + - deepseek + - ds4 + - gguf + - llm + - chat + overrides: + backend: ds4 + parameters: + model: ds4flash.gguf + files: + - filename: ds4flash.gguf + sha256: edabc92af63ad8b139f00087fbfc10a4072f37b7597f4fd9ad1dfa6f83002396 + uri: huggingface://antirez/deepseek-v4-gguf/DeepSeek-V4-Flash-Layers37-42Q4KExperts-OtherExpertLayersIQ2XXSGateUp-Q2KDown-AProjQ8-SExpQ8-OutQ8-chat-v2-imatrix-fixed.gguf +- name: deepseek-v4-flash-q4-ssd + description: | + DeepSeek V4 Flash (full 4-bit experts GGUF, ~153 GB) - only loadable via the + ds4 backend, with SSD streaming enabled so it runs on a 128 GB machine even + though the weights do not fit in RAM: routed MoE experts stream from the GGUF + on SSD while the non-routed weights stay resident. SSD streaming is Metal + (Darwin) only; generation speed depends on SSD speed and the expert cache. + Tune the routed-expert cache with the 'ssd_streaming_cache_experts:NGB' + option (default: automatic budget). See https://github.com/antirez/ds4. + urls: + - https://huggingface.co/antirez/deepseek-v4-gguf + tags: + - deepseek + - ds4 + - gguf + - llm + - chat + overrides: + backend: ds4 + options: + - ssd_streaming + parameters: + model: ds4flash.gguf + files: + - filename: ds4flash.gguf + sha256: 39e5de72ac544fdd5ffaf83ec28e36aaf3341b145235488e67d59400bbb3af55 + uri: huggingface://antirez/deepseek-v4-gguf/DeepSeek-V4-Flash-Q4KExperts-F16HC-F16Compressor-F16Indexer-Q8Attn-Q8Shared-Q8Out-chat-v2.gguf +- name: deepseek-v4-flash-q2-mtp + description: | + DeepSeek V4 Flash (IQ2XXS GGUF, ~81 GB) paired with the optional MTP + speculative-decoding weights (~3.5 GB) for a slight speedup. Only loadable + via the ds4 backend; requires >=128 GB RAM. MTP helps only with greedy + decoding (temperature 0), so the override pins temperature to 0. Metal + (Darwin) or CUDA (Linux). See https://github.com/antirez/ds4 for details. + urls: + - https://huggingface.co/antirez/deepseek-v4-gguf + tags: + - deepseek + - ds4 + - gguf + - llm + - chat + overrides: + backend: ds4 + options: + - mtp_path:ds4flash-mtp.gguf + - mtp_draft:2 + parameters: + model: ds4flash.gguf + temperature: 0 + files: + - filename: ds4flash.gguf + sha256: 31598c67c8b8744d3bcebcd19aa62253c6dc43cef3b8adf9f593656c9e86fd8c + uri: huggingface://antirez/deepseek-v4-gguf/DeepSeek-V4-Flash-IQ2XXS-w2Q2K-AProjQ8-SExpQ8-OutQ8-chat-v2.gguf + - filename: ds4flash-mtp.gguf + sha256: afd481ee689dce9037f70f39085fcdae5a5b096d521cdad43b19fa52bf8f4083 + uri: huggingface://antirez/deepseek-v4-gguf/DeepSeek-V4-Flash-MTP-Q4K-Q8_0-F32.gguf +- name: deepseek-v4-pro-q2-ssd + description: | + DeepSeek V4 Pro (IQ2XXS GGUF, ~433 GB, imatrix-tuned) - only loadable via the + ds4 backend, with SSD streaming so the Pro-class model can be run on a 128 GB + machine. This is experimental and slow: it needs ~433 GB of free SSD plus + enough RAM for the resident weights, KV cache, and routed-expert cache, and + is best used with thinking off for inspection or occasional work. SSD + streaming is Metal (Darwin) only. See https://github.com/antirez/ds4. + urls: + - https://huggingface.co/antirez/deepseek-v4-gguf + tags: + - deepseek + - ds4 + - gguf + - llm + - chat + overrides: + backend: ds4 + options: + - ssd_streaming + parameters: + model: ds4pro.gguf + files: + - filename: ds4pro.gguf + sha256: a0314d9c0e16122cd60071079124a2d17185d317c55a8f95ecb3ed3506278a96 + uri: huggingface://antirez/deepseek-v4-gguf/DeepSeek-V4-Pro-IQ2XXS-w2Q2K-AProjQ8-SExpQ8-OutQ8-Instruct-imatrix.gguf - name: parakeet-cpp-tdt_ctc-110m url: github:mudler/LocalAI/gallery/virtual.yaml@master urls: