diff --git a/docs/maintainer/resource-scheduling-and-context-cache.md b/docs/maintainer/resource-scheduling-and-context-cache.md index 5543200b96..7f0f95f2ce 100644 --- a/docs/maintainer/resource-scheduling-and-context-cache.md +++ b/docs/maintainer/resource-scheduling-and-context-cache.md @@ -510,8 +510,8 @@ candidates:全部 tools 之后、连续 leading System/Developer 之后,以 因此每个请求最多七个 prepared candidates。这个固定上限不是启动配置。 Shared catalog 是 Engine-wide 公共容量,不是每条 lineage 的配额。启用 context cache 时,默认 logical -capacity 同时覆盖 active concurrency 下限和单请求最多四个显式 markers,即 -`max(max_concurrency, kMaximumExplicitPromptCacheMarkers)`;显式配置仍完整覆盖默认值。这个下限允许较早的 +capacity 同时覆盖 active concurrency 下限和单请求最多七个 prepared candidates,即 +`max(max_concurrency, kMaximumPreparedPromptCacheCandidatesPerRequest)`;显式配置仍完整覆盖默认值。这个下限允许较早的 稳定层与较晚的滚动 marker 同时成为 owner,但是否 capture、保留或替换仍只由通用 portfolio/pressure planning 决定,不提供 Claude、compact 或 token-position 特例。 diff --git a/docs/serving.md b/docs/serving.md index 329763fe1b..7c20649bef 100644 --- a/docs/serving.md +++ b/docs/serving.md @@ -785,7 +785,7 @@ The table lists executable defaults. The startup example selects a long-context | `--host-state-slots N` | pinned Host StateImage capacity | `8` | | `--host-kv-mib N` | shared pinned Host Main/Backend KV byte capacity in MiB | `8192` | | `--max-private-continuations N` | private continuation descriptor capacity | `2 * max-concurrency` | -| `--max-shared-prefixes N` | Engine-wide shared stable-prefix descriptor capacity | `max(max-concurrency, 4)` | +| `--max-shared-prefixes N` | Engine-wide shared stable-prefix descriptor capacity | `max(max-concurrency, 7)` | | `--max-long-anchors-per-continuation N` | private long-anchor limit per continuation | `2` | | `--no-thinking` | disable thinking by default | thinking on | | `--preserve-thinking` | preserve closed-turn assistant reasoning by default | off | diff --git a/include/ninfer/types.h b/include/ninfer/types.h index a2e8b75481..8c967d82bb 100644 --- a/include/ninfer/types.h +++ b/include/ninfer/types.h @@ -20,6 +20,9 @@ using TokenId = std::int32_t; inline constexpr std::uint32_t kMaximumConcurrency = 8; inline constexpr std::size_t kMaximumContextCacheSessionKeyBytes = 256; inline constexpr std::size_t kMaximumExplicitPromptCacheMarkers = 4; +// Explicit markers plus the engine's automatic tool/leading-instruction/full-prompt candidates; +// one request's shared-prefix opportunities never exceed this (frontend.cpp opportunities.reserve). +inline constexpr std::size_t kMaximumPreparedPromptCacheCandidatesPerRequest = 7; // Aggregate encoded image/video payload retained by one prompt, independent of item count. inline constexpr std::size_t kMaximumPromptMediaBytes = 256ULL << 20; inline constexpr std::size_t kDefaultMediaCacheBytes = 1ULL << 30; @@ -127,7 +130,7 @@ struct StartupObserver { struct ContextCacheOptions { // Engine resolves every optional once at construction. With C=max_concurrency, the enabled - // defaults are H=C, R=8, Host KV=8 GiB, P=2C, S=max(C,4) and L=2; + // defaults are H=C, R=8, Host KV=8 GiB, P=2C, S=max(C,7) and L=2; // Engine::options() returns those effective values. bool enabled = true; // Extra Device checkpoint StateImage slots H. Total Device StateImage capacity is C + H. diff --git a/src/runtime/engine/model_instance.cpp b/src/runtime/engine/model_instance.cpp index e606d9a371..7a935b3716 100644 --- a/src/runtime/engine/model_instance.cpp +++ b/src/runtime/engine/model_instance.cpp @@ -112,8 +112,8 @@ EngineOptions normalize_engine_options(EngineOptions options) { const std::uint64_t default_private = 2ULL * concurrency; cache.max_private_continuations = cache.max_private_continuations.value_or(static_cast(default_private)); - cache.max_shared_prefixes = cache.max_shared_prefixes.value_or( - std::max(concurrency, static_cast(kMaximumExplicitPromptCacheMarkers))); + cache.max_shared_prefixes = cache.max_shared_prefixes.value_or(std::max( + concurrency, static_cast(kMaximumPreparedPromptCacheCandidatesPerRequest))); cache.max_long_anchors_per_continuation = cache.max_long_anchors_per_continuation.value_or(2U); if (*cache.max_private_continuations < concurrency) { diff --git a/tests/cmake/RuntimeTests.cmake b/tests/cmake/RuntimeTests.cmake index a68c1aa9fc..2f3ba5d3ca 100644 --- a/tests/cmake/RuntimeTests.cmake +++ b/tests/cmake/RuntimeTests.cmake @@ -10,6 +10,10 @@ ninfer_add_test(ninfer_resource_manager_test SOURCES "${CMAKE_CURRENT_LIST_DIR}/ ninfer_add_test(ninfer_kv_capacity_test SOURCES "${CMAKE_CURRENT_LIST_DIR}/../test_kv_capacity.cpp" LIBRARIES ninfer_runtime_support) +ninfer_add_test(ninfer_context_cache_defaults_test + SOURCES "${CMAKE_CURRENT_LIST_DIR}/../test_context_cache_defaults.cpp" + LIBRARIES ninfer_engine ninfer_core ninfer::json) + ninfer_add_test(ninfer_sampling_defaults_test SOURCES "${CMAKE_CURRENT_LIST_DIR}/../test_sampling_defaults.cpp" LIBRARIES ninfer_engine ninfer_core) diff --git a/tests/test_context_cache_defaults.cpp b/tests/test_context_cache_defaults.cpp new file mode 100644 index 0000000000..0a02f7681c --- /dev/null +++ b/tests/test_context_cache_defaults.cpp @@ -0,0 +1,63 @@ +#include "runtime/engine/model_instance.h" + +#include + +namespace { + +int check(bool condition, const char* message) { + if (condition) { return 0; } + std::cerr << message << '\n'; + return 1; +} + +} // namespace + +int main() { + using ninfer::EngineOptions; + using ninfer::kMaximumPreparedPromptCacheCandidatesPerRequest; + using ninfer::runtime::normalize_engine_options; + + int failures = 0; + + // A single request can produce up to kMaximumPreparedPromptCacheCandidatesPerRequest distinct + // shared-prefix candidates (frontend.cpp's opportunities.reserve(7U): four explicit markers + // plus the engine's tool/leading-instruction/full-prompt automatic candidates). The default + // Engine-wide shared catalog must be able to hold at least one request's own candidates even + // at the smallest concurrency, or ordinary DefaultAutomatic-evidence traffic starts losing + // cache hits to its own prior turns as soon as the catalog fills. + for (const std::uint32_t concurrency : {1U, 2U, 8U}) { + EngineOptions options; + options.max_concurrency = concurrency; + const EngineOptions normalized = normalize_engine_options(options); + const std::uint32_t default_shared_prefixes = *normalized.context_cache.max_shared_prefixes; + failures += check( + default_shared_prefixes >= + static_cast(kMaximumPreparedPromptCacheCandidatesPerRequest), + "default shared-prefix catalog capacity is smaller than one request's own candidate ceiling"); + failures += check(default_shared_prefixes >= concurrency, + "default shared-prefix catalog capacity did not cover active concurrency"); + } + + // An explicit override is still respected verbatim, including a deliberately small value. + { + EngineOptions options; + options.max_concurrency = 1; + options.context_cache.max_shared_prefixes = 1; + const EngineOptions normalized = normalize_engine_options(options); + failures += check(*normalized.context_cache.max_shared_prefixes == 1, + "explicit max_shared_prefixes override was not preserved"); + } + + // A disabled context cache still normalizes to a root-only zero capacity. + { + EngineOptions options; + options.max_concurrency = 1; + options.context_cache.enabled = false; + const EngineOptions normalized = normalize_engine_options(options); + failures += check(*normalized.context_cache.max_shared_prefixes == 0, + "disabled context cache did not normalize shared-prefix capacity to zero"); + } + + if (failures == 0) { std::cout << "ok\n"; } + return failures == 0 ? 0 : 1; +}