Skip to content

Commit bda6da7

Browse files
committed
fix: frame JSONL index reads on LF, not str.splitlines()
Four readers of the per-Goal run index frame it with `str.splitlines()` (`history.py::repair_index_duplicates`, `run_index_rebuild.py::read_index_rows`, `pending_intent.py::_actual_work_window`, `monitor_poll.py::_find_monitor_poll_turn`). The index is written with `json.dumps(..., ensure_ascii=False)`, which keeps U+0085, U+2028 and U+2029 inside a value verbatim — and `str.splitlines()` treats all three as line breaks. A record holding one in a text field arrives as two fragments, both fail `json.loads`, and the row is silently dropped. Frame on LF instead. A trailing `\r` from a CRLF file stays harmless because JSON treats it as whitespace. Reproduced before the change: one record containing U+0085 yields zero parsed rows; after it yields the record intact. The new test fails on `read_index_rows` when only that function is reverted to `splitlines()` and passes with the fix. Validation: `pytest -q tests/control_plane/test_run_index_jsonl_framing.py` -> 2 passed; `ruff check` on the five files reports `All checks passed!`. Signed-off-by: kokokoXUY <13682395396@163.com> Rebased onto current main. Rebased onto current main. Rebased edit-on-current-main: the previous head copied whole files from an older base and reverted upstream changes in them; this head applies only our edit to current main.
1 parent b9a34c3 commit bda6da7

5 files changed

Lines changed: 44 additions & 4 deletions

File tree

‎loopx/capabilities/periodic_report/pending_intent.py‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -760,7 +760,7 @@ def _actual_work_window(
760760
run_index = runtime_root / "goals" / goal_id / "runs" / "index.jsonl"
761761
if run_index.is_file():
762762
try:
763-
rows = run_index.read_text(encoding="utf-8").splitlines()
763+
rows = run_index.read_text(encoding="utf-8").split("\n")
764764
except OSError:
765765
rows = []
766766
for raw_row in rows:

‎loopx/control_plane/quota/monitor_poll.py‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -387,7 +387,7 @@ def _find_monitor_poll_turn(
387387
normalized_todo_id = normalize_todo_id(todo_id) if todo_id else None
388388
normalized_target_key = str(target_key or "").strip() or None
389389
try:
390-
lines = index_path.read_text(encoding="utf-8").splitlines()
390+
lines = index_path.read_text(encoding="utf-8").split("\n")
391391
except OSError:
392392
return None
393393
for line in reversed(lines):

‎loopx/control_plane/runtime/run_index_rebuild.py‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -38,7 +38,7 @@ def _event_identity(record: dict[str, Any]) -> dict[str, Any]:
3838

3939

4040
def read_index_rows(index_path: Path) -> tuple[list[str], list[tuple[int, dict[str, Any]]]]:
41-
raw_lines = index_path.read_text(encoding="utf-8").splitlines()
41+
raw_lines = index_path.read_text(encoding="utf-8").split("\n")
4242
rows: list[tuple[int, dict[str, Any]]] = []
4343
for line_number, line in enumerate(raw_lines, start=1):
4444
if not line.strip():

‎loopx/history.py‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -725,7 +725,7 @@ def repair_index_duplicates(
725725
else nullcontext()
726726
)
727727
with lock:
728-
raw_lines = index_path.read_text(encoding="utf-8").splitlines()
728+
raw_lines = index_path.read_text(encoding="utf-8").split("\n")
729729
grouped: dict[tuple[str, str, str], list[tuple[int, dict[str, Any]]]] = {}
730730
for line_number, line in enumerate(raw_lines, start=1):
731731
if not line.strip():
Lines changed: 40 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,40 @@
1+
"""Record framing for the JSONL run index."""
2+
3+
from __future__ import annotations
4+
5+
import json
6+
from pathlib import Path
7+
8+
from loopx.control_plane.runtime.run_index_rebuild import read_index_rows
9+
10+
NEL = "\x85"
11+
12+
13+
def test_run_index_record_with_a_raw_separator_stays_one_record(tmp_path: Path) -> None:
14+
"""`json.dumps(..., ensure_ascii=False)` keeps U+0085 verbatim in a value.
15+
16+
`str.splitlines()` treats U+0085, U+2028 and U+2029 as line breaks, so the
17+
record arrived as two fragments that both failed to parse and the row was
18+
dropped instead of being read.
19+
"""
20+
21+
index = tmp_path / "run_index.jsonl"
22+
record = {"agent_id": "agent-a", "text": f"note{NEL}with-nel"}
23+
index.write_text(json.dumps(record, ensure_ascii=False) + "\n", encoding="utf-8")
24+
25+
_, rows = read_index_rows(index)
26+
27+
assert [row for _, row in rows] == [record]
28+
29+
30+
def test_run_index_ordinary_records_are_unaffected(tmp_path: Path) -> None:
31+
index = tmp_path / "run_index.jsonl"
32+
records = [{"agent_id": "agent-a", "index": value} for value in range(3)]
33+
index.write_text(
34+
"".join(json.dumps(row, ensure_ascii=False) + "\n" for row in records),
35+
encoding="utf-8",
36+
)
37+
38+
_, rows = read_index_rows(index)
39+
40+
assert [row for _, row in rows] == records

0 commit comments

Comments
 (0)