diff --git a/examples/loopx-chat-agent-smoke.py b/examples/loopx-chat-agent-smoke.py index c14609ce8..83a98662a 100644 --- a/examples/loopx-chat-agent-smoke.py +++ b/examples/loopx-chat-agent-smoke.py @@ -18,7 +18,7 @@ if str(REPO_ROOT) not in sys.path: sys.path.insert(0, str(REPO_ROOT)) -from loopx.chat import VisibleResponseStreamFilter # noqa: E402 +from loopx.chat import CHAT_REVIEW_OPEN_TAG, VisibleResponseStreamFilter, redact_local_paths # noqa: E402 from loopx.chat_agent import ( # noqa: E402 CodexChatAgentError, CodexChatAgentSession, @@ -247,7 +247,46 @@ def get(self, *, timeout: float) -> dict[str, object]: assert chinese_early, "Chinese prose must stream without whitespace or sentence punctuation" assert chinese_early + chinese_filter.finish() == long_chinese + english_filter = VisibleResponseStreamFilter() + english_chunks = ["Version 1.2 is ready", ".", " See example.com for notes! Next,", " run the check? Done"] + english_visible = [english_filter.feed(chunk) for chunk in english_chunks] + assert english_visible[0] == "", "a decimal or version point must not end a sentence" + assert english_visible[1] == "", "sentence punctuation waits for the following whitespace" + assert english_visible[2] == "Version 1.2 is ready. See example.com for notes! ", english_visible + assert english_visible[3] == "Next, run the check? ", english_visible + assert "".join(english_visible) + english_filter.finish() == "".join(english_chunks) + + # An early sentence end must not hold back a long tail in the same chunk: + # the tail still streams through the length fallback before the provider + # sends anything else. + tail_filter = VisibleResponseStreamFilter() + tail_chunk = "Ready. " + "word " * 100 + tail_visible = tail_filter.feed(tail_chunk) + assert tail_visible.startswith("Ready. word "), tail_visible[:40] + assert len(tail_chunk) - len(tail_visible) < 160, len(tail_visible) + assert tail_visible + tail_filter.finish() == tail_chunk + protected_path = "/home/example/project" + tail_path_filter = VisibleResponseStreamFilter(protected_paths=[protected_path]) + tail_path_text = "Ready. " + "word " * 40 + f"see {protected_path}/notes.txt " + "word " * 40 + tail_path_chunks = [tail_path_text, CHAT_REVIEW_OPEN_TAG + '{"hidden": true}'] + tail_path_early = tail_path_filter.feed(tail_path_chunks[0]) + assert len(tail_path_text) - len(tail_path_early) < 160, len(tail_path_early) + tail_path_visible = tail_path_early + tail_path_filter.feed(tail_path_chunks[1]) + tail_path_filter.finish() + assert protected_path not in tail_path_visible, tail_path_visible + assert "hidden" not in tail_path_visible, tail_path_visible + assert tail_path_visible == redact_local_paths(tail_path_text, protected_paths=[protected_path]), tail_path_visible + + english_path_filter = VisibleResponseStreamFilter(protected_paths=[protected_path]) + english_path_text = f"The report is in {protected_path}/notes.txt. Review it next. " + english_path_visible = "".join( + english_path_filter.feed(english_path_text[index : index + 7]) + for index in range(0, len(english_path_text), 7) + ) + english_path_filter.finish() + assert protected_path not in english_path_visible, english_path_visible + # Splitting at sentences must not change what redaction produces. + assert english_path_visible == redact_local_paths(english_path_text, protected_paths=[protected_path]), english_path_visible + path_filter = VisibleResponseStreamFilter(protected_paths=[protected_path]) path_text = ("前" * 150) + protected_path + "/secret.txt 后续内容" path_early = "".join( diff --git a/loopx/chat.py b/loopx/chat.py index e85fa386a..c43732d39 100644 --- a/loopx/chat.py +++ b/loopx/chat.py @@ -106,6 +106,10 @@ class VisibleResponseStreamFilter: """Stream safe operator text while withholding the structured review envelope.""" _FLUSH_BOUNDARIES = {"\n", "。", "!", "?"} + # Latin sentence punctuation ends a sentence only before whitespace, so + # decimals, versions, file names and URLs never split. The split lands on + # whitespace, which the length fallback below already treats as safe. + _SPACED_SENTENCE_ENDINGS = {".", "!", "?"} _MAX_PENDING_CHARS = 160 def __init__(self, *, protected_paths: Iterable[Path | str] = ()) -> None: @@ -114,19 +118,19 @@ def __init__(self, *, protected_paths: Iterable[Path | str] = ()) -> None: self.visible_pending = "" self.envelope_started = False - def _accept_visible(self, text: str, *, final: bool) -> str: - self.visible_pending += text - if final: - ready = self.visible_pending - self.visible_pending = "" - return redact_local_paths(ready, protected_paths=self.protected_paths) + def _next_boundary(self, pending: str) -> int: boundary = -1 - search_limit = min(len(self.visible_pending), self._MAX_PENDING_CHARS) - for index, character in enumerate(self.visible_pending[:search_limit]): + search_limit = min(len(pending), self._MAX_PENDING_CHARS) + for index, character in enumerate(pending[:search_limit]): if character in self._FLUSH_BOUNDARIES: boundary = index + 1 - if boundary < 0 and len(self.visible_pending) >= self._MAX_PENDING_CHARS: - prefix = self.visible_pending[: self._MAX_PENDING_CHARS + 1] + elif ( + character in self._SPACED_SENTENCE_ENDINGS + and pending[index + 1 : index + 2] in {" ", "\t"} + ): + boundary = index + 2 + if boundary < 0 and len(pending) >= self._MAX_PENDING_CHARS: + prefix = pending[: self._MAX_PENDING_CHARS + 1] whitespace = max(prefix.rfind(" "), prefix.rfind("\t")) if whitespace >= 0: boundary = whitespace + 1 @@ -134,16 +138,30 @@ def _accept_visible(self, text: str, *, final: bool) -> str: boundary = self._MAX_PENDING_CHARS else: for index, character in enumerate( - self.visible_pending[self._MAX_PENDING_CHARS :], + pending[self._MAX_PENDING_CHARS :], start=self._MAX_PENDING_CHARS, ): if character in " \t\r\n`'\"<>": boundary = index + 1 break - if boundary < 0: + return boundary + + def _accept_visible(self, text: str, *, final: bool) -> str: + self.visible_pending += text + if final: + ready = self.visible_pending + self.visible_pending = "" + return redact_local_paths(ready, protected_paths=self.protected_paths) + # One chunk can hold several safe boundaries. Keep cutting until none + # is left, so an early sentence never holds back a long tail that the + # length fallback would otherwise release. + ready_length = 0 + while (boundary := self._next_boundary(self.visible_pending[ready_length:])) > 0: + ready_length += boundary + if not ready_length: return "" - ready = self.visible_pending[:boundary] - self.visible_pending = self.visible_pending[boundary:] + ready = self.visible_pending[:ready_length] + self.visible_pending = self.visible_pending[ready_length:] return redact_local_paths(ready, protected_paths=self.protected_paths) def feed(self, chunk: str) -> str: