Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions docs/maintainer/resource-scheduling-and-context-cache.md
Original file line number Diff line number Diff line change
Expand Up @@ -510,8 +510,8 @@ candidates:全部 tools 之后、连续 leading System/Developer 之后,以
因此每个请求最多七个 prepared candidates。这个固定上限不是启动配置。

Shared catalog 是 Engine-wide 公共容量,不是每条 lineage 的配额。启用 context cache 时,默认 logical
capacity 同时覆盖 active concurrency 下限和单请求最多四个显式 markers,即
`max(max_concurrency, kMaximumExplicitPromptCacheMarkers)`;显式配置仍完整覆盖默认值。这个下限允许较早的
capacity 同时覆盖 active concurrency 下限和单请求最多七个 prepared candidates,即
`max(max_concurrency, kMaximumPreparedPromptCacheCandidatesPerRequest)`;显式配置仍完整覆盖默认值。这个下限允许较早的
稳定层与较晚的滚动 marker 同时成为 owner,但是否 capture、保留或替换仍只由通用 portfolio/pressure
planning 决定,不提供 Claude、compact 或 token-position 特例。

Expand Down
2 changes: 1 addition & 1 deletion docs/serving.md
Original file line number Diff line number Diff line change
Expand Up @@ -785,7 +785,7 @@ The table lists executable defaults. The startup example selects a long-context
| `--host-state-slots N` | pinned Host StateImage capacity | `8` |
| `--host-kv-mib N` | shared pinned Host Main/Backend KV byte capacity in MiB | `8192` |
| `--max-private-continuations N` | private continuation descriptor capacity | `2 * max-concurrency` |
| `--max-shared-prefixes N` | Engine-wide shared stable-prefix descriptor capacity | `max(max-concurrency, 4)` |
| `--max-shared-prefixes N` | Engine-wide shared stable-prefix descriptor capacity | `max(max-concurrency, 7)` |
| `--max-long-anchors-per-continuation N` | private long-anchor limit per continuation | `2` |
| `--no-thinking` | disable thinking by default | thinking on |
| `--preserve-thinking` | preserve closed-turn assistant reasoning by default | off |
Expand Down
5 changes: 4 additions & 1 deletion include/ninfer/types.h
Original file line number Diff line number Diff line change
Expand Up @@ -20,6 +20,9 @@ using TokenId = std::int32_t;
inline constexpr std::uint32_t kMaximumConcurrency = 8;
inline constexpr std::size_t kMaximumContextCacheSessionKeyBytes = 256;
inline constexpr std::size_t kMaximumExplicitPromptCacheMarkers = 4;
// Explicit markers plus the engine's automatic tool/leading-instruction/full-prompt candidates;
// one request's shared-prefix opportunities never exceed this (frontend.cpp opportunities.reserve).
inline constexpr std::size_t kMaximumPreparedPromptCacheCandidatesPerRequest = 7;
// Aggregate encoded image/video payload retained by one prompt, independent of item count.
inline constexpr std::size_t kMaximumPromptMediaBytes = 256ULL << 20;
inline constexpr std::size_t kDefaultMediaCacheBytes = 1ULL << 30;
Expand Down Expand Up @@ -127,7 +130,7 @@ struct StartupObserver {

struct ContextCacheOptions {
// Engine resolves every optional once at construction. With C=max_concurrency, the enabled
// defaults are H=C, R=8, Host KV=8 GiB, P=2C, S=max(C,4) and L=2;
// defaults are H=C, R=8, Host KV=8 GiB, P=2C, S=max(C,7) and L=2;
// Engine::options() returns those effective values.
bool enabled = true;
// Extra Device checkpoint StateImage slots H. Total Device StateImage capacity is C + H.
Expand Down
4 changes: 2 additions & 2 deletions src/runtime/engine/model_instance.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -112,8 +112,8 @@ EngineOptions normalize_engine_options(EngineOptions options) {
const std::uint64_t default_private = 2ULL * concurrency;
cache.max_private_continuations =
cache.max_private_continuations.value_or(static_cast<std::uint32_t>(default_private));
cache.max_shared_prefixes = cache.max_shared_prefixes.value_or(
std::max(concurrency, static_cast<std::uint32_t>(kMaximumExplicitPromptCacheMarkers)));
cache.max_shared_prefixes = cache.max_shared_prefixes.value_or(std::max(
concurrency, static_cast<std::uint32_t>(kMaximumPreparedPromptCacheCandidatesPerRequest)));
cache.max_long_anchors_per_continuation = cache.max_long_anchors_per_continuation.value_or(2U);

if (*cache.max_private_continuations < concurrency) {
Expand Down
4 changes: 4 additions & 0 deletions tests/cmake/RuntimeTests.cmake
Original file line number Diff line number Diff line change
Expand Up @@ -10,6 +10,10 @@ ninfer_add_test(ninfer_resource_manager_test SOURCES "${CMAKE_CURRENT_LIST_DIR}/
ninfer_add_test(ninfer_kv_capacity_test SOURCES "${CMAKE_CURRENT_LIST_DIR}/../test_kv_capacity.cpp"
LIBRARIES ninfer_runtime_support)

ninfer_add_test(ninfer_context_cache_defaults_test
SOURCES "${CMAKE_CURRENT_LIST_DIR}/../test_context_cache_defaults.cpp"
LIBRARIES ninfer_engine ninfer_core ninfer::json)

ninfer_add_test(ninfer_sampling_defaults_test
SOURCES "${CMAKE_CURRENT_LIST_DIR}/../test_sampling_defaults.cpp"
LIBRARIES ninfer_engine ninfer_core)
63 changes: 63 additions & 0 deletions tests/test_context_cache_defaults.cpp
Original file line number Diff line number Diff line change
@@ -0,0 +1,63 @@
#include "runtime/engine/model_instance.h"

#include <iostream>

namespace {

int check(bool condition, const char* message) {
if (condition) { return 0; }
std::cerr << message << '\n';
return 1;
}

} // namespace

int main() {
using ninfer::EngineOptions;
using ninfer::kMaximumPreparedPromptCacheCandidatesPerRequest;
using ninfer::runtime::normalize_engine_options;

int failures = 0;

// A single request can produce up to kMaximumPreparedPromptCacheCandidatesPerRequest distinct
// shared-prefix candidates (frontend.cpp's opportunities.reserve(7U): four explicit markers
// plus the engine's tool/leading-instruction/full-prompt automatic candidates). The default
// Engine-wide shared catalog must be able to hold at least one request's own candidates even
// at the smallest concurrency, or ordinary DefaultAutomatic-evidence traffic starts losing
// cache hits to its own prior turns as soon as the catalog fills.
for (const std::uint32_t concurrency : {1U, 2U, 8U}) {
EngineOptions options;
options.max_concurrency = concurrency;
const EngineOptions normalized = normalize_engine_options(options);
const std::uint32_t default_shared_prefixes = *normalized.context_cache.max_shared_prefixes;
failures += check(
default_shared_prefixes >=
static_cast<std::uint32_t>(kMaximumPreparedPromptCacheCandidatesPerRequest),
"default shared-prefix catalog capacity is smaller than one request's own candidate ceiling");
failures += check(default_shared_prefixes >= concurrency,
"default shared-prefix catalog capacity did not cover active concurrency");
}

// An explicit override is still respected verbatim, including a deliberately small value.
{
EngineOptions options;
options.max_concurrency = 1;
options.context_cache.max_shared_prefixes = 1;
const EngineOptions normalized = normalize_engine_options(options);
failures += check(*normalized.context_cache.max_shared_prefixes == 1,
"explicit max_shared_prefixes override was not preserved");
}

// A disabled context cache still normalizes to a root-only zero capacity.
{
EngineOptions options;
options.max_concurrency = 1;
options.context_cache.enabled = false;
const EngineOptions normalized = normalize_engine_options(options);
failures += check(*normalized.context_cache.max_shared_prefixes == 0,
"disabled context cache did not normalize shared-prefix capacity to zero");
}

if (failures == 0) { std::cout << "ok\n"; }
return failures == 0 ? 0 : 1;
}