From a042a8d755b314d1a84d6c612ec8143285ff219d Mon Sep 17 00:00:00 2001 From: kokokoXUY <13682395396@163.com> Date: Sun, 27 Sep 2026 11:05:13 +0800 Subject: [PATCH] fix: frame the remaining JSONL readers on LF #5117 fixed four readers of the per-Goal run index. The same framing defect exists in nine more places that parse one JSON document per line after `read_text(...).splitlines()`: - `chat_store.py` (stored chat rows) - `doctor.py` (installation index) - `domain_state.py` (domain state rows) - `event_sourced_state.py` (the append-only event log) - `domain_packs/issue_fix.py` (two readers) - `capabilities/explore/result_log.py` (three readers) All of these files also write with `json.dumps(..., ensure_ascii=False)`, which leaves U+0085/U+2028/U+2029 in a value verbatim, and `str.splitlines()` treats them as line breaks: one record becomes two fragments, `json.loads` fails on both, and the row is dropped or reported as invalid. Frame on LF instead. Validation: a new case in `tests/test_event_sourced_state_store.py` writes one event whose title carries U+0085 and asserts it round-trips; it raises `StateEventError` before the change and passes after. The other eight sites are the same one-line shape. `pytest -q tests/test_event_sourced_state_store.py tests/test_chat_store_input_validation.py` -> 23 passed, 1 failed, where that failure is `test_failure_before_replace_leaves_old_stream_intact` injecting `OSError("injected pre-publication failure")` inside the file lock; it fails identically with these edits reverted, so it is pre-existing on this host. `ruff check` on the changed files reports `All checks passed!`. Signed-off-by: kokokoXUY <13682395396@163.com> Keep the test imports above the module constant (E402). --- loopx/capabilities/explore/result_log.py | 6 ++--- loopx/chat_store.py | 2 +- loopx/doctor.py | 2 +- loopx/domain_packs/issue_fix.py | 4 ++-- loopx/domain_state.py | 2 +- loopx/event_sourced_state.py | 2 +- tests/test_event_sourced_state_store.py | 29 ++++++++++++++++++++++++ 7 files changed, 38 insertions(+), 9 deletions(-) diff --git a/loopx/capabilities/explore/result_log.py b/loopx/capabilities/explore/result_log.py index 6f5dfc0c1c..433fce8d3c 100644 --- a/loopx/capabilities/explore/result_log.py +++ b/loopx/capabilities/explore/result_log.py @@ -506,7 +506,7 @@ def append_explore_result_events( with exclusive_file_lock(log_path): existing_by_id: dict[str, dict[str, Any]] = {} if log_path.exists(): - for line in log_path.read_text(encoding="utf-8").splitlines(): + for line in log_path.read_text(encoding="utf-8").split("\n"): try: current = json.loads(line) except json.JSONDecodeError: @@ -546,7 +546,7 @@ def load_explore_result_events( if not log_path.exists(): return [] events: list[dict[str, Any]] = [] - for line in log_path.read_text(encoding="utf-8").splitlines(): + for line in log_path.read_text(encoding="utf-8").split("\n"): stripped = line.strip() if not stripped: continue @@ -579,7 +579,7 @@ def load_explore_result_events_strict( if not log_path.exists(): return [] events: list[dict[str, Any]] = [] - for line_number, line in enumerate(log_path.read_text(encoding="utf-8").splitlines(), start=1): + for line_number, line in enumerate(log_path.read_text(encoding="utf-8").split("\n"), start=1): stripped = line.strip() if not stripped: continue diff --git a/loopx/chat_store.py b/loopx/chat_store.py index 259aa22fc2..a182ed84c5 100644 --- a/loopx/chat_store.py +++ b/loopx/chat_store.py @@ -75,7 +75,7 @@ def _read_json(path: Path) -> dict[str, Any]: def _read_jsonl(path: Path) -> list[dict[str, Any]]: try: - lines = path.read_text(encoding="utf-8").splitlines() + lines = path.read_text(encoding="utf-8").split("\n") except OSError: return [] rows: list[dict[str, Any]] = [] diff --git a/loopx/doctor.py b/loopx/doctor.py index bffb9fc22c..ab59235440 100644 --- a/loopx/doctor.py +++ b/loopx/doctor.py @@ -625,7 +625,7 @@ def latest_promotion_readiness_event(runtime_root: Path, goal_id: str | None = N ) for index_path, current_goal_id, source in indexes: try: - lines = index_path.read_text(encoding="utf-8").splitlines() + lines = index_path.read_text(encoding="utf-8").split("\n") except OSError: continue for line in lines: diff --git a/loopx/domain_packs/issue_fix.py b/loopx/domain_packs/issue_fix.py index 89e9f9742f..2a847f37f9 100644 --- a/loopx/domain_packs/issue_fix.py +++ b/loopx/domain_packs/issue_fix.py @@ -170,7 +170,7 @@ def promote_issue_fix_feasibility_ledger_jsonl( canonical_existing: dict[str, Any] | None = None if path.exists(): for index, line in enumerate( - path.read_text(encoding="utf-8").splitlines(), start=1 + path.read_text(encoding="utf-8").split("\n"), start=1 ): if not line.strip(): continue @@ -530,7 +530,7 @@ def retain_issue_fix_repository_snapshot_jsonl( path = Path(ledger_path) existing_rows: list[dict[str, Any]] = [] if path.exists(): - for line in path.read_text(encoding="utf-8").splitlines(): + for line in path.read_text(encoding="utf-8").split("\n"): try: value = json.loads(line) except (TypeError, ValueError): diff --git a/loopx/domain_state.py b/loopx/domain_state.py index a447a2bc93..3590e6739c 100644 --- a/loopx/domain_state.py +++ b/loopx/domain_state.py @@ -71,7 +71,7 @@ def upsert_domain_state_jsonl( candidate = {**payload, "domain_state_key": key} if path.exists(): for index, line in enumerate( - path.read_text(encoding="utf-8").splitlines(), start=1 + path.read_text(encoding="utf-8").split("\n"), start=1 ): if not line.strip(): continue diff --git a/loopx/event_sourced_state.py b/loopx/event_sourced_state.py index 82f97dffc0..e33293dde1 100644 --- a/loopx/event_sourced_state.py +++ b/loopx/event_sourced_state.py @@ -578,7 +578,7 @@ class AppendOnlyStateEventStore: def load(self) -> list[dict[str, Any]]: events: list[dict[str, Any]] = [] if self.path.exists(): - for line_number, line in enumerate(self.path.read_text(encoding="utf-8").splitlines(), start=1): + for line_number, line in enumerate(self.path.read_text(encoding="utf-8").split("\n"), start=1): if not line.strip(): continue try: diff --git a/tests/test_event_sourced_state_store.py b/tests/test_event_sourced_state_store.py index e09bd8c903..86346bdc1f 100644 --- a/tests/test_event_sourced_state_store.py +++ b/tests/test_event_sourced_state_store.py @@ -12,6 +12,8 @@ make_state_event, ) +NEL = "\x85" + def test_load_observes_events_appended_by_another_store(tmp_path: Path) -> None: event_log = tmp_path / "events.jsonl" @@ -429,3 +431,30 @@ def test_concurrent_processes_publish_contiguous_batches(tmp_path: Path) -> None if row["event_id"].startswith(f"{worker}-") ] assert positions == list(range(positions[0], positions[0] + 3)) + + +def test_load_reads_a_record_whose_value_carries_u0085(tmp_path: Path) -> None: + """The event log is one JSON document per LF. + + `json.dumps(..., ensure_ascii=False)` keeps U+0085 inside a value verbatim, + and `str.splitlines()` treats it as a line break, so a valid event arrived + as two fragments and `load()` raised `StateEventError`. + """ + + event_log = tmp_path / "events.jsonl" + store = AppendOnlyStateEventStore(event_log) + event = make_state_event( + event_id="evt-raw-separator", + goal_id="goal-a", + event_type=TODO_ADDED, + refs={"todo_id": "todo_raw_separator"}, + payload={"role": "agent", "title": f"Observe the durable event.{NEL}"}, + recorded_at="2026-09-06T00:00:00Z", + ) + event_log.write_text( + json.dumps(event, ensure_ascii=False) + "\n", encoding="utf-8" + ) + + loaded = store.load() + + assert [item["event_id"] for item in loaded] == ["evt-raw-separator"]