From b6a21c99e7289f7d4ed8b9f24003fadd5d94bedf Mon Sep 17 00:00:00 2001 From: AutoDev Bot Date: Thu, 1 Oct 2026 00:09:05 +0800 Subject: [PATCH 1/2] fix(reference_utils): guard split_continuous_references against plain bracketed prose split_continuous_references previously rewrote any single [...] block that contained a comma into pseudo reference tags, so ordinary prose like "The set is [apple, banana] here" was corrupted into "The set is [apple][banana] here". Add a shape guard: only split when every comma-separated item in the block matches : (the actual reference tag shape, e.g. [1:92ff35fb, 4:bfe6f044]); otherwise return the text unchanged. Adds tests/mem_os/test_reference_utils.py covering the regression case, mixed / non-reference items, and boundary conditions (empty, multiple brackets, single item). Fixes #2446 --- src/memos/mem_os/utils/reference_utils.py | 29 ++++++++-- tests/mem_os/test_reference_utils.py | 69 +++++++++++++++++++++++ 2 files changed, 93 insertions(+), 5 deletions(-) create mode 100644 tests/mem_os/test_reference_utils.py diff --git a/src/memos/mem_os/utils/reference_utils.py b/src/memos/mem_os/utils/reference_utils.py index 09b812207..7c1365345 100644 --- a/src/memos/mem_os/utils/reference_utils.py +++ b/src/memos/mem_os/utils/reference_utils.py @@ -1,18 +1,31 @@ +import re + from memos.memories.textual.item import ( TextualMemoryItem, ) +# A single reference item inside a bracketed list looks like ``:``, +# e.g. ``1:92ff35fb``. The id part is a hex-looking memory id, so we only +# require it to be non-empty and free of commas / brackets. Surrounding +# whitespace on either the ref id or the memory id is tolerated. +_REFERENCE_ITEM_RE = re.compile(r"^\s*\d+:[^,\[\]]+\s*$") + + def split_continuous_references(text: str) -> str: """ Split continuous reference tags into individual reference tags. - Converts patterns like [1:92ff35fb, 4:bfe6f044] to [1:92ff35fb] [4:bfe6f044] + Converts patterns like [1:92ff35fb, 4:bfe6f044] to [1:92ff35fb][4:bfe6f044]. - Only processes text if: - 1. '[' appears exactly once - 2. ']' appears exactly once - 3. Contains commas between '[' and ']' + Only processes text if all of the following hold: + + 1. ``[`` appears exactly once + 2. ``]`` appears exactly once + 3. There is at least one comma between the brackets + 4. Every comma-separated item inside the brackets matches ``:`` + (the reference tag shape). If any item does not, the block is treated + as ordinary prose (e.g. ``[apple, banana]``) and returned unchanged. Args: text (str): Text containing reference tags @@ -41,6 +54,12 @@ def split_continuous_references(text: str) -> str: # Check if there's a comma between brackets if "," not in content_between_brackets: return text + # Shape guard: only rewrite when every item looks like a reference tag + # (``:``). Otherwise the block is plain bracketed prose and + # must be preserved verbatim (issue #2446). + items = content_between_brackets.split(",") + if not all(_REFERENCE_ITEM_RE.match(item) for item in items): + return text text = text.replace(content_between_brackets, content_between_brackets.replace(", ", "][")) text = text.replace(content_between_brackets, content_between_brackets.replace(",", "][")) diff --git a/tests/mem_os/test_reference_utils.py b/tests/mem_os/test_reference_utils.py new file mode 100644 index 000000000..4cddc4b2b --- /dev/null +++ b/tests/mem_os/test_reference_utils.py @@ -0,0 +1,69 @@ +"""Tests for src/memos/mem_os/utils/reference_utils.py. + +Focus: split_continuous_references shape guard. +Related issue: #2446 — plain bracketed prose such as "[apple, banana]" must +not be rewritten as "[apple][banana]". Only reference lists whose items are +``int:`` pairs (e.g. ``[1:92ff35fb, 4:bfe6f044]``) should be split. +""" + +from memos.mem_os.utils.reference_utils import split_continuous_references + + +class TestSplitContinuousReferences: + """Shape guard on ``split_continuous_references``.""" + + # --- happy path: real reference lists still split ------------------------ + + def test_splits_two_int_id_items(self): + assert ( + split_continuous_references("See [1:92ff35fb, 4:bfe6f044] now") + == "See [1:92ff35fb][4:bfe6f044] now" + ) + + def test_splits_three_int_id_items(self): + assert split_continuous_references("[1:aa, 2:bb, 3:cc]") == "[1:aa][2:bb][3:cc]" + + def test_handles_comma_without_space(self): + assert ( + split_continuous_references("prefix [1:aa,2:bb] suffix") == "prefix [1:aa][2:bb] suffix" + ) + + # --- shape guard: non-reference brackets stay untouched ------------------ + + def test_plain_prose_list_untouched(self): + """The regression case from issue #2446.""" + text = "The set is [apple, banana] here" + assert split_continuous_references(text) == text + + def test_mixed_reference_and_prose_untouched(self): + """If even one item is not int: the whole block is preserved.""" + text = "Look at [1:92ff35fb, banana] please" + assert split_continuous_references(text) == text + + def test_numeric_only_items_untouched(self): + """Numbers without a colon are not references.""" + text = "Pick [1, 2, 3] please" + assert split_continuous_references(text) == text + + def test_non_integer_prefix_untouched(self): + """The item prefix must be a decimal integer.""" + text = "Combine [a:1, b:2]" + assert split_continuous_references(text) == text + + # --- boundary conditions unchanged --------------------------------------- + + def test_empty_string_returns_empty(self): + assert split_continuous_references("") == "" + + def test_no_brackets_returns_text_unchanged(self): + text = "no brackets, just commas" + assert split_continuous_references(text) == text + + def test_multiple_open_brackets_returns_unchanged(self): + text = "many [1:aa, 2:bb] and [3:cc, 4:dd]" + assert split_continuous_references(text) == text + + def test_single_reference_item_unchanged(self): + """A single item has no comma so nothing to split.""" + text = "just [1:aa]" + assert split_continuous_references(text) == text From 907c95e7a14a56feb5b724e16e00b299e699a314 Mon Sep 17 00:00:00 2001 From: AutoDev Bot Date: Thu, 1 Oct 2026 00:22:56 +0800 Subject: [PATCH 2/2] fix(reference_utils): split mixed comma separators in one pass The previous two-step str.replace approach in split_continuous_references was broken for blocks that mixed ", " and bare "," separators: Input: "[1:aa, 2:bb,3:cc]" content_between_brackets = "1:aa, 2:bb,3:cc" After the first replace, text became "[1:aa][2:bb,3:cc]"; the second replace searched for the original substring, could no longer find it, and the bare comma between 2:bb and 3:cc was never split. Rebuild the bracketed block in a single pass by joining the already- computed items with "][" and stripping surrounding whitespace, so all separator styles collapse identically. Adds two regression tests for mixed separator styles (PR #2450 review). --- src/memos/mem_os/utils/reference_utils.py | 12 ++++++++---- tests/mem_os/test_reference_utils.py | 18 ++++++++++++++++++ 2 files changed, 26 insertions(+), 4 deletions(-) diff --git a/src/memos/mem_os/utils/reference_utils.py b/src/memos/mem_os/utils/reference_utils.py index 7c1365345..fa002ee63 100644 --- a/src/memos/mem_os/utils/reference_utils.py +++ b/src/memos/mem_os/utils/reference_utils.py @@ -60,10 +60,14 @@ def split_continuous_references(text: str) -> str: items = content_between_brackets.split(",") if not all(_REFERENCE_ITEM_RE.match(item) for item in items): return text - text = text.replace(content_between_brackets, content_between_brackets.replace(", ", "][")) - text = text.replace(content_between_brackets, content_between_brackets.replace(",", "][")) - - return text + # Rebuild the bracketed block in a single pass so mixed separator styles + # (``", "`` and bare ``","`` in the same block) are all split correctly. + # The previous two-step ``str.replace`` approach was broken: after the + # first pass rewrote ``", "`` occurrences, the original substring no + # longer existed in ``text`` and the second pass never fired, leaving + # bare commas unsplit (PR #2450 review). + joined = "][".join(item.strip() for item in items) + return text[: open_bracket_pos + 1] + joined + text[close_bracket_pos:] def process_streaming_references_complete(text_buffer: str) -> tuple[str, str]: diff --git a/tests/mem_os/test_reference_utils.py b/tests/mem_os/test_reference_utils.py index 4cddc4b2b..00c28cd4d 100644 --- a/tests/mem_os/test_reference_utils.py +++ b/tests/mem_os/test_reference_utils.py @@ -28,6 +28,24 @@ def test_handles_comma_without_space(self): split_continuous_references("prefix [1:aa,2:bb] suffix") == "prefix [1:aa][2:bb] suffix" ) + def test_handles_mixed_separator_styles(self): + """Regression: mixed ``", "`` and bare ``","`` separators in the same + block must all be split (PR #2450 review). + + The previous two-step ``str.replace`` implementation left the bare + comma between ``2:bb`` and ``3:cc`` intact, producing + ``"[1:aa][2:bb,3:cc]"``. + """ + assert ( + split_continuous_references("[1:aa, 2:bb,3:cc]") == "[1:aa][2:bb][3:cc]" + ) + + def test_handles_mixed_separator_styles_reversed(self): + """Bare comma first, then ``", "`` — symmetric to the case above.""" + assert ( + split_continuous_references("[1:aa,2:bb, 3:cc]") == "[1:aa][2:bb][3:cc]" + ) + # --- shape guard: non-reference brackets stay untouched ------------------ def test_plain_prose_list_untouched(self):