Skip to content

Commit 41a68ac

Browse files
Sebastian BraunCopilot
andcommitted
fix(agent): retry summary-rewrite with summary as source before giving up
known_targets_msg (the whitelist of every existing concept/entity page) grows with the KB, same as concepts-plan's brief lists (#226) - a document whose concepts-plan call still fit can push summary-rewrite over the context window once that whitelist gets large enough. Previously any exception from this call, including litellm.ContextWindowExceededError, fell straight through to the v1-summary fallback, silently forfeiting the improved cross-linking summary-rewrite exists to produce. Mirrors the existing concepts-plan fix: on _NON_RETRYABLE_LLM_ERRORS, retry once with doc_msg dropped (the rewrite prompt only needs summary_msg and known_targets_msg, not the original document) before falling through to the unchanged v1 fallback. Resolves #256 Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com>
1 parent 55807a4 commit 41a68ac

2 files changed

Lines changed: 160 additions & 12 deletions

File tree

‎openkb/agent/compiler.py‎

Lines changed: 39 additions & 12 deletions
Original file line numberDiff line numberDiff line change
@@ -2365,21 +2365,48 @@ async def _gen_entity_update(ent: dict) -> tuple[str, str, str, str]:
23652365
# the full whitelist, so the summary is always written and never wiped.
23662366
if rewrite_summary:
23672367
candidate: str | None = None
2368+
summary_rewrite_user_msg = {"role": "user", "content": _SUMMARY_REWRITE_USER}
23682369
try:
23692370
# No max_tokens cap — matches the v1 summary call. The rewrite
23702371
# prompt asks the model to keep length within ±20% of the v1.
2371-
rewrite_raw = _llm_call(
2372-
model,
2373-
[
2374-
system_msg,
2375-
doc_msg, # cached (BP1)
2376-
summary_msg, # cached (BP2) — contains the v1 summary text
2377-
known_targets_msg, # cached (BP3) — whitelist
2378-
{"role": "user", "content": _SUMMARY_REWRITE_USER},
2379-
],
2380-
"summary-rewrite",
2381-
bundle=bundle,
2382-
)
2372+
try:
2373+
rewrite_raw = _llm_call(
2374+
model,
2375+
[
2376+
system_msg,
2377+
doc_msg, # cached (BP1)
2378+
summary_msg, # cached (BP2) — contains the v1 summary text
2379+
known_targets_msg, # cached (BP3) — whitelist
2380+
summary_rewrite_user_msg,
2381+
],
2382+
"summary-rewrite",
2383+
bundle=bundle,
2384+
)
2385+
except _NON_RETRYABLE_LLM_ERRORS as exc:
2386+
# known_targets_msg grows with the KB just like the
2387+
# concepts-plan index (#226) — a doc whose earlier calls fit
2388+
# can still blow the window here once the whitelist gets
2389+
# large enough. Retry once without the full document: the
2390+
# rewrite prompt only asks the model to reconcile the
2391+
# already-generated summary (summary_msg) against the
2392+
# whitelist, not the original document.
2393+
logger.warning(
2394+
"summary-rewrite exceeded context window for %s: %s; "
2395+
"retrying with summary as source",
2396+
doc_name,
2397+
exc,
2398+
)
2399+
sys.stdout.write(
2400+
f" [WARN] summary-rewrite exceeded context window for {doc_name} — "
2401+
"retrying with the summary as source instead of the full document.\n"
2402+
)
2403+
sys.stdout.flush()
2404+
rewrite_raw = _llm_call(
2405+
model,
2406+
[system_msg, summary_msg, known_targets_msg, summary_rewrite_user_msg],
2407+
"summary-rewrite",
2408+
bundle=bundle,
2409+
)
23832410
candidate = rewrite_raw.strip()
23842411
# Strip frontmatter if the model added one anyway.
23852412
cand_parts = frontmatter.split(candidate)

‎tests/test_compiler.py‎

Lines changed: 121 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -1616,6 +1616,127 @@ def side_effect(*args, **kwargs):
16161616
index_text = (wiki / "index.md").read_text()
16171617
assert "[[concepts/transformer]]" in index_text
16181618

1619+
@pytest.mark.asyncio
1620+
async def test_summary_rewrite_context_window_exceeded_retries_with_summary_only(
1621+
self, tmp_path
1622+
):
1623+
"""summary-rewrite's prompt carries doc_msg plus the whitelist of
1624+
every existing concept/entity page (known_targets_msg), which grows
1625+
with the KB (#226) just like the concepts-plan index — so it can
1626+
blow the context window on a document whose concepts-plan call (a
1627+
smaller prompt, no whitelist) still fit. Mirrors the concepts-plan
1628+
fix: retry once with doc_msg dropped before falling back to v1."""
1629+
wiki, source_path = self._setup_kb(tmp_path)
1630+
(wiki / "concepts" / "transformer.md").write_text(
1631+
"---\ndescription: Existing\n---\n\nExisting content.", encoding="utf-8"
1632+
)
1633+
1634+
v1_summary_content = "# Summary\n\nDiscusses transformers."
1635+
summary_response = json.dumps(
1636+
{"description": "A real summary", "content": v1_summary_content}
1637+
)
1638+
plan_response = json.dumps(
1639+
{
1640+
"create": [],
1641+
"update": [{"name": "transformer", "title": "Transformer"}],
1642+
"related": [],
1643+
}
1644+
)
1645+
concept_response = json.dumps({"description": "C", "content": "# T\n\nUpdated body."})
1646+
rewritten_summary = "# Summary\n\nRewritten: discusses [[concepts/transformer]]."
1647+
error = litellm.ContextWindowExceededError(
1648+
message="prompt is too long: 200400 tokens > 200000 maximum",
1649+
model="claude-sonnet-4-5",
1650+
llm_provider="anthropic",
1651+
)
1652+
call_count = {"n": 0}
1653+
1654+
def side_effect(*args, **kwargs):
1655+
idx = call_count["n"]
1656+
call_count["n"] += 1
1657+
if idx == 0:
1658+
return [_mock_response(summary_response)]
1659+
if idx == 1:
1660+
return [_mock_response(plan_response)]
1661+
if idx == 2:
1662+
raise error
1663+
return [_mock_response(rewritten_summary)]
1664+
1665+
with patch("openkb.agent.compiler.litellm") as mock_litellm:
1666+
mock_litellm.completion = MagicMock(side_effect=side_effect)
1667+
mock_litellm.acompletion = AsyncMock(side_effect=_mock_acompletion([concept_response]))
1668+
await compile_short_doc("doc", source_path, tmp_path, "claude-sonnet-4-5")
1669+
# summary + concepts-plan + summary-rewrite (full doc, fails) +
1670+
# summary-rewrite retry (summary only, succeeds).
1671+
assert mock_litellm.completion.call_count == 4
1672+
1673+
# The retry attempt must not still include the full document message
1674+
# (_SUMMARY_USER's "Full text:" marker only ever appears in doc_msg).
1675+
retry_messages = mock_litellm.completion.call_args_list[3].kwargs["messages"]
1676+
retry_text = json.dumps(retry_messages)
1677+
assert "Full text:" not in retry_text
1678+
1679+
summary_path = wiki / "summaries" / "doc.md"
1680+
assert summary_path.exists()
1681+
text = summary_path.read_text()
1682+
assert "Rewritten" in text # retried rewrite content used, not v1
1683+
assert "[[concepts/transformer]]" in text
1684+
1685+
@pytest.mark.asyncio
1686+
async def test_summary_rewrite_context_window_retry_also_fails_falls_back_to_v1(self, tmp_path):
1687+
"""If the summary-only retry ALSO hits the context window (or any
1688+
other error), the existing v1 fallback still applies unchanged —
1689+
the real v1 summary (ghost-stripped) is written, never lost."""
1690+
wiki, source_path = self._setup_kb(tmp_path)
1691+
(wiki / "concepts" / "transformer.md").write_text(
1692+
"---\ndescription: Existing\n---\n\nExisting content.", encoding="utf-8"
1693+
)
1694+
1695+
v1_summary_content = (
1696+
"# Summary\n\nDiscusses [[concepts/transformer]] and [[concepts/ghost]]."
1697+
)
1698+
summary_response = json.dumps(
1699+
{"description": "A real summary", "content": v1_summary_content}
1700+
)
1701+
plan_response = json.dumps(
1702+
{
1703+
"create": [],
1704+
"update": [{"name": "transformer", "title": "Transformer"}],
1705+
"related": [],
1706+
}
1707+
)
1708+
concept_response = json.dumps({"description": "C", "content": "# T\n\nUpdated body."})
1709+
error = litellm.ContextWindowExceededError(
1710+
message="prompt is too long: 200400 tokens > 200000 maximum",
1711+
model="claude-sonnet-4-5",
1712+
llm_provider="anthropic",
1713+
)
1714+
call_count = {"n": 0}
1715+
1716+
def side_effect(*args, **kwargs):
1717+
idx = call_count["n"]
1718+
call_count["n"] += 1
1719+
if idx == 0:
1720+
return [_mock_response(summary_response)]
1721+
if idx == 1:
1722+
return [_mock_response(plan_response)]
1723+
raise error # both the first rewrite attempt and its retry fail
1724+
1725+
with patch("openkb.agent.compiler.litellm") as mock_litellm:
1726+
mock_litellm.completion = MagicMock(side_effect=side_effect)
1727+
mock_litellm.acompletion = AsyncMock(side_effect=_mock_acompletion([concept_response]))
1728+
# Must not raise out of compile_short_doc.
1729+
await compile_short_doc("doc", source_path, tmp_path, "claude-sonnet-4-5")
1730+
assert mock_litellm.completion.call_count == 4
1731+
1732+
summary_path = wiki / "summaries" / "doc.md"
1733+
assert summary_path.exists()
1734+
text = summary_path.read_text()
1735+
assert "Discusses" in text # real v1 content kept, not lost
1736+
assert "[[concepts/transformer]]" in text # valid link kept
1737+
assert "[[concepts/ghost]]" not in text # ghost link stripped
1738+
assert "ghost" in text # display text preserved
1739+
16191740

16201741
class TestOversizedDocumentSkip:
16211742
"""A document whose content can't fit an LLM call is treated like an

0 commit comments

Comments
 (0)