From 1e440956f8b4e05f0639160fd3359e6f57506edb Mon Sep 17 00:00:00 2001 From: simpleqt <89645338+simpleqt@users.noreply.github.com> Date: Mon, 28 Sep 2026 11:20:06 +0800 Subject: [PATCH 1/2] fix(mem_os): split references on mixed comma styles split_continuous_references applied two sequential str.replace passes, one for ', ' and one for ','. The first matching pass rewrote the content substring, so the second pass found nothing: a tag mixing both styles ('[a,b, c]') kept its earlier references merged on the streaming chat path. Split on every comma with re.sub(r',\\s*', ...)' instead. Red/green verified: the mixed-separator test fails against the old implementation and passes with the fix. --- src/memos/mem_os/utils/reference_utils.py | 14 +++++-- tests/mem_os/test_reference_utils.py | 49 +++++++++++++++++++++++ 2 files changed, 59 insertions(+), 4 deletions(-) create mode 100644 tests/mem_os/test_reference_utils.py diff --git a/src/memos/mem_os/utils/reference_utils.py b/src/memos/mem_os/utils/reference_utils.py index 09b812207..1e83f2544 100644 --- a/src/memos/mem_os/utils/reference_utils.py +++ b/src/memos/mem_os/utils/reference_utils.py @@ -1,3 +1,5 @@ +import re + from memos.memories.textual.item import ( TextualMemoryItem, ) @@ -41,8 +43,14 @@ def split_continuous_references(text: str) -> str: # Check if there's a comma between brackets if "," not in content_between_brackets: return text - text = text.replace(content_between_brackets, content_between_brackets.replace(", ", "][")) - text = text.replace(content_between_brackets, content_between_brackets.replace(",", "][")) + # Split on every comma regardless of the whitespace that follows it: LLM + # output mixes "a, b" and "a,b" freely, and the previous two-pass + # str.replace handled only one style per call, leaving earlier references + # merged when both styles appeared in the same tag. + text = text.replace( + content_between_brackets, + re.sub(r",\s*", "][", content_between_brackets), + ) return text @@ -57,8 +65,6 @@ def process_streaming_references_complete(text_buffer: str) -> tuple[str, str]: Returns: tuple[str, str]: (processed_text, remaining_buffer) """ - import re - # Pattern to match complete reference tags: [refid:memoriesID] complete_pattern = r"\[\d+:[^\]]+\]" diff --git a/tests/mem_os/test_reference_utils.py b/tests/mem_os/test_reference_utils.py new file mode 100644 index 000000000..9ecedb0b0 --- /dev/null +++ b/tests/mem_os/test_reference_utils.py @@ -0,0 +1,49 @@ +import unittest + +from memos.mem_os.utils.reference_utils import split_continuous_references + + +class TestSplitContinuousReferences(unittest.TestCase): + def test_spaced_comma(self): + self.assertEqual( + split_continuous_references("[1:92ff35fb, 4:bfe6f044]"), + "[1:92ff35fb][4:bfe6f044]", + ) + + def test_tight_comma(self): + self.assertEqual( + split_continuous_references("[1:92ff35fb,4:bfe6f044]"), + "[1:92ff35fb][4:bfe6f044]", + ) + + def test_mixed_separators(self): + """Mixed ', ' and ',' separators used to leave earlier refs merged. + + The old two-pass str.replace applied one separator style per call; + the first successful pass removed the substring the second pass + searched for, so with "[a,b, c]" only the last boundary was split. + """ + self.assertEqual( + split_continuous_references("[1:92ff35fb,4:bfe6f044, 7:abcd1234]"), + "[1:92ff35fb][4:bfe6f044][7:abcd1234]", + ) + + def test_surrounding_text_preserved(self): + self.assertEqual( + split_continuous_references("see refs [1:aaaa, 2:bbbb] for details"), + "see refs [1:aaaa][2:bbbb] for details", + ) + + def test_non_reference_text_untouched(self): + for text in ( + "", + "plain text", + "[no commas here]", + "two [brackets] twice [here]", + "backwards ]here[", + ): + self.assertEqual(split_continuous_references(text), text) + + +if __name__ == "__main__": + unittest.main() From 66137f4045bf0b3b39ec6adfdc98c2c7aa53a208 Mon Sep 17 00:00:00 2001 From: simpleqt <89645338+simpleqt@users.noreply.github.com> Date: Mon, 28 Sep 2026 14:48:54 +0800 Subject: [PATCH 2/2] fix(mem_os): only split reference-shaped tags split_continuous_references rewrote ANY single bracketed comma list on the streaming chat path: mathematical or plain text like 'interval [x, y]' came out as 'interval [x][y]'. Restrict the split to content with the reference-tag shape (a numeric id before a colon in every element), which is the documented input format. Red/green verified: the new test fails against the old implementation ([x, y] rewritten to [x][y]) and passes with the fix. Signed-off-by: simpleqt <89645338+simpleqt@users.noreply.github.com> --- src/memos/mem_os/utils/reference_utils.py | 5 +++++ tests/mem_os/test_reference_utils.py | 9 +++++++++ 2 files changed, 14 insertions(+) diff --git a/src/memos/mem_os/utils/reference_utils.py b/src/memos/mem_os/utils/reference_utils.py index 1e83f2544..a843cf024 100644 --- a/src/memos/mem_os/utils/reference_utils.py +++ b/src/memos/mem_os/utils/reference_utils.py @@ -43,6 +43,11 @@ def split_continuous_references(text: str) -> str: # Check if there's a comma between brackets if "," not in content_between_brackets: return text + # Only reference tags (a numeric id before a colon in every element) are + # split; ordinary bracketed text such as "[x, y]" must pass through + # untouched. + if not re.fullmatch(r"\s*\d+:[^,\s]+(?:,\s*\d+:[^,\s]+)*\s*", content_between_brackets): + return text # Split on every comma regardless of the whitespace that follows it: LLM # output mixes "a, b" and "a,b" freely, and the previous two-pass # str.replace handled only one style per call, leaving earlier references diff --git a/tests/mem_os/test_reference_utils.py b/tests/mem_os/test_reference_utils.py index 9ecedb0b0..c1c400b6a 100644 --- a/tests/mem_os/test_reference_utils.py +++ b/tests/mem_os/test_reference_utils.py @@ -28,6 +28,15 @@ def test_mixed_separators(self): "[1:92ff35fb][4:bfe6f044][7:abcd1234]", ) + def test_ordinary_bracketed_text_untouched(self): + # "[x, y]" has no reference-tag shape (numeric id + colon); it must + # not be rewritten into "[x][y]" on the streaming chat path + self.assertEqual( + split_continuous_references("interval [x, y] ends"), + "interval [x, y] ends", + ) + self.assertEqual(split_continuous_references("cite [1, 2] here"), "cite [1, 2] here") + def test_surrounding_text_preserved(self): self.assertEqual( split_continuous_references("see refs [1:aaaa, 2:bbbb] for details"),