Skip to content

Commit e05b0dc

Browse files
committed
fix: frame JSONL index reads on LF, not str.splitlines()
Four readers of the per-Goal run index frame it with `str.splitlines()` (`history.py::repair_index_duplicates`, `run_index_rebuild.py::read_index_rows`, `pending_intent.py::_actual_work_window`, `monitor_poll.py::_find_monitor_poll_turn`). The index is written with `json.dumps(..., ensure_ascii=False)`, which keeps U+0085, U+2028 and U+2029 inside a value verbatim — and `str.splitlines()` treats all three as line breaks. A record holding one in a text field arrives as two fragments, both fail `json.loads`, and the row is silently dropped. Frame on LF instead. A trailing `\r` from a CRLF file stays harmless because JSON treats it as whitespace. Reproduced before the change: one record containing U+0085 yields zero parsed rows; after it yields the record intact. The new test fails on `read_index_rows` when only that function is reverted to `splitlines()` and passes with the fix. Validation: `pytest -q tests/control_plane/test_run_index_jsonl_framing.py` -> 2 passed; `ruff check` on the five files reports `All checks passed!`. Signed-off-by: kokokoXUY <13682395396@163.com> Rebased onto current main.
1 parent 03b7e66 commit e05b0dc

5 files changed

Lines changed: 52 additions & 16 deletions

File tree

‎loopx/capabilities/periodic_report/pending_intent.py‎

Lines changed: 2 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -760,7 +760,8 @@ def _actual_work_window(
760760
run_index = runtime_root / "goals" / goal_id / "runs" / "index.jsonl"
761761
if run_index.is_file():
762762
try:
763-
rows = run_index.read_text(encoding="utf-8").splitlines()
763+
# LF framing, not `splitlines()`: see `loopx/history.py` for the same reason.
764+
rows = run_index.read_text(encoding="utf-8").split("\n")
764765
except OSError:
765766
rows = []
766767
for raw_row in rows:

‎loopx/control_plane/quota/monitor_poll.py‎

Lines changed: 2 additions & 13 deletions
Original file line numberDiff line numberDiff line change
@@ -387,7 +387,8 @@ def _find_monitor_poll_turn(
387387
normalized_todo_id = normalize_todo_id(todo_id) if todo_id else None
388388
normalized_target_key = str(target_key or "").strip() or None
389389
try:
390-
lines = index_path.read_text(encoding="utf-8").splitlines()
390+
# LF framing keeps one record one record when a value carries U+0085.
391+
lines = index_path.read_text(encoding="utf-8").split("\n")
391392
except OSError:
392393
return None
393394
for line in reversed(lines):
@@ -698,7 +699,6 @@ def record_quota_monitor_poll_for_decision(
698699
task_lease_idempotency_key: str | None = None,
699700
task_lease_expected_version: int | None = None,
700701
use_current_task_lease: bool = False,
701-
auxiliary_settlement_todo: Mapping[str, Any] | None = None,
702702
turn_instance_id: str | None = None,
703703
_index_lock_held: bool = False,
704704
status_reloader: Callable[[], dict[str, Any]] | None = None,
@@ -751,17 +751,6 @@ def record_quota_monitor_poll_for_decision(
751751
registry_path=registry_path,
752752
runtime_root=runtime_root,
753753
)
754-
if auxiliary_settlement_todo is not None:
755-
decision["auxiliary_settlement_todo"] = {
756-
key: auxiliary_settlement_todo.get(key)
757-
for key in (
758-
"todo_id",
759-
"task_class",
760-
"status",
761-
"claimed_by",
762-
"excluded_agents",
763-
)
764-
}
765754
observation = _observation_packet(
766755
before=before,
767756
agent_id=agent_id,

‎loopx/control_plane/runtime/run_index_rebuild.py‎

Lines changed: 3 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -38,7 +38,9 @@ def _event_identity(record: dict[str, Any]) -> dict[str, Any]:
3838

3939

4040
def read_index_rows(index_path: Path) -> tuple[list[str], list[tuple[int, dict[str, Any]]]]:
41-
raw_lines = index_path.read_text(encoding="utf-8").splitlines()
41+
# One JSON document per LF: the writer keeps non-ASCII verbatim, so a value
42+
# carrying U+0085 must not be treated as a line break here.
43+
raw_lines = index_path.read_text(encoding="utf-8").split("\n")
4244
rows: list[tuple[int, dict[str, Any]]] = []
4345
for line_number, line in enumerate(raw_lines, start=1):
4446
if not line.strip():

‎loopx/history.py‎

Lines changed: 5 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -699,7 +699,11 @@ def repair_index_duplicates(
699699
else nullcontext()
700700
)
701701
with lock:
702-
raw_lines = index_path.read_text(encoding="utf-8").splitlines()
702+
# The index is one JSON document per LF. `json.dumps(..., ensure_ascii=False)`
703+
# leaves U+0085/U+2028/U+2029 in a value verbatim, and `str.splitlines()`
704+
# treats those as line breaks, which tore one record into two unparsable
705+
# fragments. Frame on LF instead.
706+
raw_lines = index_path.read_text(encoding="utf-8").split("\n")
703707
grouped: dict[tuple[str, str, str], list[tuple[int, dict[str, Any]]]] = {}
704708
for line_number, line in enumerate(raw_lines, start=1):
705709
if not line.strip():
Lines changed: 40 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,40 @@
1+
"""Record framing for the JSONL run index."""
2+
3+
from __future__ import annotations
4+
5+
import json
6+
from pathlib import Path
7+
8+
from loopx.control_plane.runtime.run_index_rebuild import read_index_rows
9+
10+
NEL = "\x85"
11+
12+
13+
def test_run_index_record_with_a_raw_separator_stays_one_record(tmp_path: Path) -> None:
14+
"""`json.dumps(..., ensure_ascii=False)` keeps U+0085 verbatim in a value.
15+
16+
`str.splitlines()` treats U+0085, U+2028 and U+2029 as line breaks, so the
17+
record arrived as two fragments that both failed to parse and the row was
18+
dropped instead of being read.
19+
"""
20+
21+
index = tmp_path / "run_index.jsonl"
22+
record = {"agent_id": "agent-a", "text": f"note{NEL}with-nel"}
23+
index.write_text(json.dumps(record, ensure_ascii=False) + "\n", encoding="utf-8")
24+
25+
_, rows = read_index_rows(index)
26+
27+
assert [row for _, row in rows] == [record]
28+
29+
30+
def test_run_index_ordinary_records_are_unaffected(tmp_path: Path) -> None:
31+
index = tmp_path / "run_index.jsonl"
32+
records = [{"agent_id": "agent-a", "index": value} for value in range(3)]
33+
index.write_text(
34+
"".join(json.dumps(row, ensure_ascii=False) + "\n" for row in records),
35+
encoding="utf-8",
36+
)
37+
38+
_, rows = read_index_rows(index)
39+
40+
assert [row for _, row in rows] == records

0 commit comments

Comments
 (0)