From e05b0dcf20812fc02136ae313558d65c59542d49 Mon Sep 17 00:00:00 2001 From: kokokoXUY <13682395396@163.com> Date: Sun, 27 Sep 2026 13:17:44 +0800 Subject: [PATCH] fix: frame JSONL index reads on LF, not str.splitlines() MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Four readers of the per-Goal run index frame it with `str.splitlines()` (`history.py::repair_index_duplicates`, `run_index_rebuild.py::read_index_rows`, `pending_intent.py::_actual_work_window`, `monitor_poll.py::_find_monitor_poll_turn`). The index is written with `json.dumps(..., ensure_ascii=False)`, which keeps U+0085, U+2028 and U+2029 inside a value verbatim — and `str.splitlines()` treats all three as line breaks. A record holding one in a text field arrives as two fragments, both fail `json.loads`, and the row is silently dropped. Frame on LF instead. A trailing `\r` from a CRLF file stays harmless because JSON treats it as whitespace. Reproduced before the change: one record containing U+0085 yields zero parsed rows; after it yields the record intact. The new test fails on `read_index_rows` when only that function is reverted to `splitlines()` and passes with the fix. Validation: `pytest -q tests/control_plane/test_run_index_jsonl_framing.py` -> 2 passed; `ruff check` on the five files reports `All checks passed!`. Signed-off-by: kokokoXUY <13682395396@163.com> Rebased onto current main. --- .../periodic_report/pending_intent.py | 3 +- loopx/control_plane/quota/monitor_poll.py | 15 +------ .../runtime/run_index_rebuild.py | 4 +- loopx/history.py | 6 ++- .../test_run_index_jsonl_framing.py | 40 +++++++++++++++++++ 5 files changed, 52 insertions(+), 16 deletions(-) create mode 100644 tests/control_plane/test_run_index_jsonl_framing.py diff --git a/loopx/capabilities/periodic_report/pending_intent.py b/loopx/capabilities/periodic_report/pending_intent.py index 57b2e4908a..5838af0c1e 100644 --- a/loopx/capabilities/periodic_report/pending_intent.py +++ b/loopx/capabilities/periodic_report/pending_intent.py @@ -760,7 +760,8 @@ def _actual_work_window( run_index = runtime_root / "goals" / goal_id / "runs" / "index.jsonl" if run_index.is_file(): try: - rows = run_index.read_text(encoding="utf-8").splitlines() + # LF framing, not `splitlines()`: see `loopx/history.py` for the same reason. + rows = run_index.read_text(encoding="utf-8").split("\n") except OSError: rows = [] for raw_row in rows: diff --git a/loopx/control_plane/quota/monitor_poll.py b/loopx/control_plane/quota/monitor_poll.py index 20a14f04aa..5899c5090f 100644 --- a/loopx/control_plane/quota/monitor_poll.py +++ b/loopx/control_plane/quota/monitor_poll.py @@ -387,7 +387,8 @@ def _find_monitor_poll_turn( normalized_todo_id = normalize_todo_id(todo_id) if todo_id else None normalized_target_key = str(target_key or "").strip() or None try: - lines = index_path.read_text(encoding="utf-8").splitlines() + # LF framing keeps one record one record when a value carries U+0085. + lines = index_path.read_text(encoding="utf-8").split("\n") except OSError: return None for line in reversed(lines): @@ -698,7 +699,6 @@ def record_quota_monitor_poll_for_decision( task_lease_idempotency_key: str | None = None, task_lease_expected_version: int | None = None, use_current_task_lease: bool = False, - auxiliary_settlement_todo: Mapping[str, Any] | None = None, turn_instance_id: str | None = None, _index_lock_held: bool = False, status_reloader: Callable[[], dict[str, Any]] | None = None, @@ -751,17 +751,6 @@ def record_quota_monitor_poll_for_decision( registry_path=registry_path, runtime_root=runtime_root, ) - if auxiliary_settlement_todo is not None: - decision["auxiliary_settlement_todo"] = { - key: auxiliary_settlement_todo.get(key) - for key in ( - "todo_id", - "task_class", - "status", - "claimed_by", - "excluded_agents", - ) - } observation = _observation_packet( before=before, agent_id=agent_id, diff --git a/loopx/control_plane/runtime/run_index_rebuild.py b/loopx/control_plane/runtime/run_index_rebuild.py index 65c765bb9b..ffa73fed0e 100644 --- a/loopx/control_plane/runtime/run_index_rebuild.py +++ b/loopx/control_plane/runtime/run_index_rebuild.py @@ -38,7 +38,9 @@ def _event_identity(record: dict[str, Any]) -> dict[str, Any]: def read_index_rows(index_path: Path) -> tuple[list[str], list[tuple[int, dict[str, Any]]]]: - raw_lines = index_path.read_text(encoding="utf-8").splitlines() + # One JSON document per LF: the writer keeps non-ASCII verbatim, so a value + # carrying U+0085 must not be treated as a line break here. + raw_lines = index_path.read_text(encoding="utf-8").split("\n") rows: list[tuple[int, dict[str, Any]]] = [] for line_number, line in enumerate(raw_lines, start=1): if not line.strip(): diff --git a/loopx/history.py b/loopx/history.py index 26e7b08f99..2ed5a90fac 100644 --- a/loopx/history.py +++ b/loopx/history.py @@ -699,7 +699,11 @@ def repair_index_duplicates( else nullcontext() ) with lock: - raw_lines = index_path.read_text(encoding="utf-8").splitlines() + # The index is one JSON document per LF. `json.dumps(..., ensure_ascii=False)` + # leaves U+0085/U+2028/U+2029 in a value verbatim, and `str.splitlines()` + # treats those as line breaks, which tore one record into two unparsable + # fragments. Frame on LF instead. + raw_lines = index_path.read_text(encoding="utf-8").split("\n") grouped: dict[tuple[str, str, str], list[tuple[int, dict[str, Any]]]] = {} for line_number, line in enumerate(raw_lines, start=1): if not line.strip(): diff --git a/tests/control_plane/test_run_index_jsonl_framing.py b/tests/control_plane/test_run_index_jsonl_framing.py new file mode 100644 index 0000000000..3b514456eb --- /dev/null +++ b/tests/control_plane/test_run_index_jsonl_framing.py @@ -0,0 +1,40 @@ +"""Record framing for the JSONL run index.""" + +from __future__ import annotations + +import json +from pathlib import Path + +from loopx.control_plane.runtime.run_index_rebuild import read_index_rows + +NEL = "\x85" + + +def test_run_index_record_with_a_raw_separator_stays_one_record(tmp_path: Path) -> None: + """`json.dumps(..., ensure_ascii=False)` keeps U+0085 verbatim in a value. + + `str.splitlines()` treats U+0085, U+2028 and U+2029 as line breaks, so the + record arrived as two fragments that both failed to parse and the row was + dropped instead of being read. + """ + + index = tmp_path / "run_index.jsonl" + record = {"agent_id": "agent-a", "text": f"note{NEL}with-nel"} + index.write_text(json.dumps(record, ensure_ascii=False) + "\n", encoding="utf-8") + + _, rows = read_index_rows(index) + + assert [row for _, row in rows] == [record] + + +def test_run_index_ordinary_records_are_unaffected(tmp_path: Path) -> None: + index = tmp_path / "run_index.jsonl" + records = [{"agent_id": "agent-a", "index": value} for value in range(3)] + index.write_text( + "".join(json.dumps(row, ensure_ascii=False) + "\n" for row in records), + encoding="utf-8", + ) + + _, rows = read_index_rows(index) + + assert [row for _, row in rows] == records