From 5555a3c5690a27f11aa41b71bfb0e9d54419e2cc Mon Sep 17 00:00:00 2001 From: Ettore Di Giacinto Date: Mon, 31 Aug 2026 22:50:48 +0200 Subject: [PATCH 1/4] fix(nodes): skip checksum sidecars when staging option dirs stageDirectory and countStageableFiles already skip them, but stageOptionDir did not - and it is the path sherpa-onnx voices take for espeak-ng-data. The receiver writes ".sha256" for every file it accepts, so staging the sidecars made it write sidecars for those in turn, one level deeper on every load. Observed on a live node: ".sha256" repeated eleven times, 5077 junk files out of 7832 in the models dir, and still growing. Staging never finished, so vits-piper-it_IT-paola-sherpa stayed permanently "staging on node" and every realtime warmup needing that voice failed with the session then going silent. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_0142UfUh8HWxdim5JZqf8Tr6 Signed-off-by: Ettore Di Giacinto --- core/services/nodes/router.go | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/core/services/nodes/router.go b/core/services/nodes/router.go index d094d0f2076e..11ec2d6c639a 100644 --- a/core/services/nodes/router.go +++ b/core/services/nodes/router.go @@ -1874,6 +1874,16 @@ func (r *SmartRouter) stageOptionDir(ctx context.Context, node *BackendNode, dir if walkErr != nil || d.IsDir() { return nil } + // Same reason as stageDirectory: the receiver writes ".sha256" for + // every file it accepts, so staging the sidecars makes it write sidecars + // for those in turn. Option dirs are walked on every load, so each pass + // added a level - an espeak-ng-data tree observed in the wild had grown + // to ".sha256" repeated eleven times and 5077 junk files, which is + // enough to keep a sherpa-onnx voice permanently "staging" and fail + // every realtime warmup that needs it. + if isHashSidecar(path) { + return nil + } if _, err := r.fileStager.EnsureRemote(ctx, node.ID, path, keyFn(path)); err != nil { xlog.Warn("Failed to stage option directory file, skipping", "path", path, "error", err) } From 475dc254be9bfdcd20986cb4e333dc0d60815421 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Tue, 1 Sep 2026 00:06:05 +0200 Subject: [PATCH 2/4] chore: :arrow_up: Update ikawrakow/ik_llama.cpp to `3c58ae373a0081c884099f435fb16ca720852bf7` (#11809) :arrow_up: Update ikawrakow/ik_llama.cpp Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- backend/cpp/ik-llama-cpp/Makefile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/cpp/ik-llama-cpp/Makefile b/backend/cpp/ik-llama-cpp/Makefile index 2ee3f30cda2a..c659e6d35e32 100644 --- a/backend/cpp/ik-llama-cpp/Makefile +++ b/backend/cpp/ik-llama-cpp/Makefile @@ -1,5 +1,5 @@ -IK_LLAMA_VERSION?=15dddc60b3fc937a9e2a210359ecce392ccdf446 +IK_LLAMA_VERSION?=3c58ae373a0081c884099f435fb16ca720852bf7 LLAMA_REPO?=https://github.com/ikawrakow/ik_llama.cpp CMAKE_ARGS?= From 2ad42384167a10a195e146fdabff8aa74533ecca Mon Sep 17 00:00:00 2001 From: Claudio Maradonna Date: Tue, 1 Sep 2026 00:06:23 +0200 Subject: [PATCH 3/4] fix(ds4): separate prefilled reasoning from content (#11802) DS4 appends the opening thinking marker to tokenizer-templated prompts, so generated text begins directly with reasoning bytes. Starting DsmlParser in TEXT therefore puts the reasoning and closing marker in visible content. Start the parser in THINK for structured chat requests with thinking enabled in both Predict and PredictStream. Keep the default TEXT state for raw prompts and reasoning-off requests, and add incremental regression coverage. Assisted-by: Codex:gpt-5 Signed-off-by: Claudio Maradonna --- backend/cpp/ds4/dsml_parser.cpp | 3 +- backend/cpp/ds4/dsml_parser.h | 6 +- backend/cpp/ds4/dsml_parser_test.cpp | 133 +++++++++++++++++++++++++++ backend/cpp/ds4/grpc-server.cpp | 16 +++- 4 files changed, 151 insertions(+), 7 deletions(-) create mode 100644 backend/cpp/ds4/dsml_parser_test.cpp diff --git a/backend/cpp/ds4/dsml_parser.cpp b/backend/cpp/ds4/dsml_parser.cpp index 6fb88a9fc610..e360c580c2a8 100644 --- a/backend/cpp/ds4/dsml_parser.cpp +++ b/backend/cpp/ds4/dsml_parser.cpp @@ -92,7 +92,8 @@ std::string json_escape(const std::string &in) { } // namespace -DsmlParser::DsmlParser() = default; +DsmlParser::DsmlParser(bool starts_in_thinking) + : state_(starts_in_thinking ? State::THINK : State::TEXT) {} bool DsmlParser::IsInDsmlStructural() const { switch (state_) { diff --git a/backend/cpp/ds4/dsml_parser.h b/backend/cpp/ds4/dsml_parser.h index c09833673bb9..14b977af68ce 100644 --- a/backend/cpp/ds4/dsml_parser.h +++ b/backend/cpp/ds4/dsml_parser.h @@ -17,7 +17,9 @@ struct ParserEvent { // Streaming parser. Stateless across instances; one per Predict call. class DsmlParser { public: - DsmlParser(); + // The chat prompt may already contain the opening thinking marker, so the + // generated text can begin directly with reasoning bytes. + explicit DsmlParser(bool starts_in_thinking = false); // Feed a chunk of raw model-emitted text. Appends classified events to // `out`. May buffer the tail of `chunk` internally if it looks like a @@ -43,7 +45,7 @@ class DsmlParser { private: enum class State { TEXT, THINK, TOOL_CALLS, INVOKE, PARAM_VALUE }; - State state_ = State::TEXT; + State state_; std::string buf_; std::string current_tool_name_; int tool_index_ = -1; diff --git a/backend/cpp/ds4/dsml_parser_test.cpp b/backend/cpp/ds4/dsml_parser_test.cpp new file mode 100644 index 000000000000..325c362524fb --- /dev/null +++ b/backend/cpp/ds4/dsml_parser_test.cpp @@ -0,0 +1,133 @@ +// SPDX-License-Identifier: MIT +// Standalone regression tests for the DSML streaming parser. +// +// The repository's backend/cpp/run-unit-tests.sh harness compiles each +// *_test.cpp as a single translation unit, so include the implementation here. + +#include "dsml_parser.cpp" + +#include +#include +#include +#include + +namespace { + +struct ParsedText { + std::string content; + std::string reasoning; +}; + +int failures = 0; + +void check_equal(const std::string &got, const std::string &want, + const char *name) { + if (got == want) return; + std::fprintf(stderr, "FAIL %s: got \"%s\", want \"%s\"\n", + name, got.c_str(), want.c_str()); + failures++; +} + +void collect_text(const std::vector &events, + ParsedText *parsed) { + for (const auto &event : events) { + if (event.type == ds4cpp::ParserEvent::CONTENT) { + parsed->content += event.text; + } else if (event.type == ds4cpp::ParserEvent::REASONING) { + parsed->reasoning += event.text; + } + } +} + +ParsedText parse_chunks(ds4cpp::DsmlParser *parser, + const std::vector &chunks) { + ParsedText parsed; + for (const auto &chunk : chunks) { + std::vector events; + parser->Feed(chunk, events); + collect_text(events, &parsed); + } + std::vector events; + parser->Flush(events); + collect_text(events, &parsed); + return parsed; +} + +template +void test_reasoning_opened_by_prompt() { + if constexpr (!std::is_constructible_v) { + std::fprintf(stderr, + "FAIL reasoning_opened_by_prompt: parser cannot start in thinking state\n"); + failures++; + } else { + Parser parser(true); + ParsedText parsed = parse_chunks( + &parser, + {"We need to calculate factorial recursively.Here is the answer."}); + check_equal(parsed.reasoning, + "We need to calculate factorial recursively.", + "reasoning_opened_by_prompt:reasoning"); + check_equal(parsed.content, "Here is the answer.", + "reasoning_opened_by_prompt:content"); + } +} + +template +Parser text_parser() { + if constexpr (std::is_constructible_v) { + return Parser(false); + } else { + return Parser(); + } +} + +void test_reasoning_disabled() { + auto parser = text_parser(); + ParsedText parsed = parse_chunks(&parser, {"Here is the answer."}); + check_equal(parsed.reasoning, "", "reasoning_disabled:reasoning"); + check_equal(parsed.content, "Here is the answer.", + "reasoning_disabled:content"); +} + +void test_explicit_think_tag() { + auto parser = text_parser(); + ParsedText parsed = parse_chunks( + &parser, {"reasoninganswer"}); + check_equal(parsed.reasoning, "reasoning", "explicit_think_tag:reasoning"); + check_equal(parsed.content, "answer", "explicit_think_tag:content"); +} + +template +void test_split_think_close_marker() { + if constexpr (!std::is_constructible_v) { + std::fprintf(stderr, + "FAIL split_think_close_marker: parser cannot start in thinking state\n"); + failures++; + } else { + Parser parser(true); + ParsedText parsed = parse_chunks( + &parser, + {"We need ", "to calculate ", "factorial", "", + "Here is ", "the answer."}); + check_equal(parsed.reasoning, "We need to calculate factorial", + "split_think_close_marker:reasoning"); + check_equal(parsed.content, "Here is the answer.", + "split_think_close_marker:content"); + } +} + +} // namespace + +int main() { + test_reasoning_opened_by_prompt(); + test_reasoning_disabled(); + test_explicit_think_tag(); + test_split_think_close_marker(); + + if (failures == 0) { + std::fprintf(stderr, "all dsml_parser checks passed\n"); + return 0; + } + std::fprintf(stderr, "%d check(s) failed\n", failures); + return 1; +} diff --git a/backend/cpp/ds4/grpc-server.cpp b/backend/cpp/ds4/grpc-server.cpp index 924da5476994..2118fd1cb270 100644 --- a/backend/cpp/ds4/grpc-server.cpp +++ b/backend/cpp/ds4/grpc-server.cpp @@ -771,7 +771,12 @@ class DS4Backend final : public backend::Backend::Service { build_prompt(g_engine, request, &prompt); int n_predict = request->tokens() > 0 ? request->tokens() : 256; - CollectCtx collect = {g_engine, "", {}, reply, 0, {}, "", ""}; + const bool think_enabled = ds4_think_mode_enabled(parse_think_mode(request)); + const bool starts_in_thinking = think_enabled && + request->usetokenizertemplate() && request->messages_size() > 0; + CollectCtx collect = { + g_engine, "", ds4cpp::DsmlParser(starts_in_thinking), + reply, 0, {}, "", ""}; std::string cache_key = render_prompt_text(request); size_t cache_hit = maybe_load_cache(cache_key); (void)cache_hit; // future: skip prompt prefix if hit covers full prompt @@ -789,7 +794,6 @@ class DS4Backend final : public backend::Backend::Service { if (rc == 0) { const int eos = ds4_token_eos(g_engine); const int draft_max = ds4_engine_mtp_draft_tokens(g_engine); - const bool think_enabled = ds4_think_mode_enabled(parse_think_mode(request)); int produced = 0; while (produced < n_predict) { SampleParams sp = compute_sample_params(request, collect.parser, think_enabled); @@ -871,7 +875,12 @@ class DS4Backend final : public backend::Backend::Service { build_prompt(g_engine, request, &prompt); int n_predict = request->tokens() > 0 ? request->tokens() : 256; - StreamCtx s = {g_engine, writer, {}, 0, false, {}}; + const bool think_enabled = ds4_think_mode_enabled(parse_think_mode(request)); + const bool starts_in_thinking = think_enabled && + request->usetokenizertemplate() && request->messages_size() > 0; + StreamCtx s = { + g_engine, writer, ds4cpp::DsmlParser(starts_in_thinking), + 0, false, {}}; std::string cache_key = render_prompt_text(request); size_t cache_hit = maybe_load_cache(cache_key); (void)cache_hit; @@ -884,7 +893,6 @@ class DS4Backend final : public backend::Backend::Service { if (rc == 0) { const int eos = ds4_token_eos(g_engine); const int draft_max = ds4_engine_mtp_draft_tokens(g_engine); - const bool think_enabled = ds4_think_mode_enabled(parse_think_mode(request)); int produced = 0; while (produced < n_predict && !s.aborted) { SampleParams sp = compute_sample_params(request, s.parser, think_enabled); From 2dcd853a2b5696cda67c469d438b733a7851110f Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Tue, 1 Sep 2026 00:06:49 +0200 Subject: [PATCH 4/4] chore: :arrow_up: Update mudler/vllm.cpp to `6a544bdb89eb5a3512ac922241439e45f24d74d4` (#11797) :arrow_up: Update mudler/vllm.cpp Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- backend/go/vllm-cpp/Makefile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/go/vllm-cpp/Makefile b/backend/go/vllm-cpp/Makefile index 1ab67324a4e3..e7888e402518 100644 --- a/backend/go/vllm-cpp/Makefile +++ b/backend/go/vllm-cpp/Makefile @@ -11,7 +11,7 @@ JOBS?=$(shell nproc --ignore=1 2>/dev/null || sysctl -n hw.ncpu 2>/dev/null || e # vllm.cpp version VLLM_CPP_REPO?=https://github.com/mudler/vllm.cpp -VLLM_CPP_VERSION?=150b37852c123f7855fb219b37347572ca9427e7 +VLLM_CPP_VERSION?=6a544bdb89eb5a3512ac922241439e45f24d74d4 # MLX GEMM provider (darwin/metal only; see the metal branch below for why). # Consumed as the prebuilt pip wheel: building MLX from source needs `xcrun