From 6c7f7aba9153f46eaacd3ab99bc08d40f5e27a88 Mon Sep 17 00:00:00 2001 From: rain <1504569896@qq.com> Date: Mon, 21 Sep 2026 17:09:13 +0000 Subject: [PATCH 1/2] fix: default BM25 tokenizer splits CJK characters for bare-term retrieval Update the default bm25_tokenization_regex of InMemoryDocumentStore so CJK scripts are tokenized per character, enabling single-character bare-term retrieval for Chinese/Japanese/Korean. Also updates to_dict expectations affected by the new default. Refinements from review: - Extract the default pattern into _CJK_CHAR_CLASS / _DEFAULT_BM25_TOKENIZATION_REGEX so the source, the to_dict tests and the docs reference a single definition instead of a duplicated literal. - Narrow the kana ranges so punctuation such as the katakana middle dot (U+30FB), the kana iteration marks and the combining voicing marks no longer become standalone tokens (and no longer inflate doc_len), while keeping the prolonged sound mark (U+30FC). - Cover the remaining CJK/Jamo blocks: halfwidth katakana (U+FF66-U+FF9F), CJK Extension A (U+3400-U+4DBF), Katakana Phonetic Extensions (U+31F0-U+31FF), CJK Compatibility Ideographs (U+F900-U+FAFF) and Hangul Jamo Extended-A/B (U+A960-U+A97F, U+D7B0-U+D7FF). - Unicode NFC-normalize the text in _tokenize_bm25 so equivalent spellings (for example decomposed vs. composed Hangul) share the same terms. - Expand the bm25_tokenization_regex docstring and update the two hand-maintained docs pages (documentcleaner, documentsplitter) to the new serialized default. --- .../preprocessors/documentcleaner.mdx | 2 +- .../preprocessors/documentsplitter.mdx | 2 +- .../in_memory/document_store.py | 36 ++++++++++++--- .../bm25-cjk-tokenization-1a2b3c4d5e6f7.yaml | 11 +++++ .../test_multi_query_embedding_retriever.py | 3 +- .../test_multi_query_text_retriever.py | 3 +- .../retrievers/test_multi_retriever.py | 3 +- .../test_sentence_window_retriever.py | 3 +- .../test_text_embedding_retriever.py | 3 +- test/document_stores/test_in_memory.py | 44 ++++++++++++++++++- test/tools/test_pipeline_tool.py | 3 +- 11 files changed, 99 insertions(+), 14 deletions(-) create mode 100644 releasenotes/notes/bm25-cjk-tokenization-1a2b3c4d5e6f7.yaml diff --git a/docs-website/docs/pipeline-components/preprocessors/documentcleaner.mdx b/docs-website/docs/pipeline-components/preprocessors/documentcleaner.mdx index 468aca6b825..100493084de 100644 --- a/docs-website/docs/pipeline-components/preprocessors/documentcleaner.mdx +++ b/docs-website/docs/pipeline-components/preprocessors/documentcleaner.mdx @@ -137,7 +137,7 @@ components: init_parameters: bm25_algorithm: BM25L bm25_parameters: {} - bm25_tokenization_regex: (?u)\\b\\w+\\b + bm25_tokenization_regex: '[^\W\u1100-\u11ff\ua960-\ua97f\ud7b0-\ud7ff\u3130-\u318f\uac00-\ud7af\u3041-\u3096\u30a1-\u30fa\u30fc\u31f0-\u31ff\u3400-\u4dbf\u4e00-\u9fff\uf900-\ufaff\uff66-\uff9f]+|[\u1100-\u11ff\ua960-\ua97f\ud7b0-\ud7ff\u3130-\u318f\uac00-\ud7af\u3041-\u3096\u30a1-\u30fa\u30fc\u31f0-\u31ff\u3400-\u4dbf\u4e00-\u9fff\uf900-\ufaff\uff66-\uff9f]' embedding_similarity_function: dot_product index: 64e4f9ab-87fb-47fd-b390-dabcfda61447 return_embedding: true diff --git a/docs-website/docs/pipeline-components/preprocessors/documentsplitter.mdx b/docs-website/docs/pipeline-components/preprocessors/documentsplitter.mdx index 720eeaa14ae..43fc9de6b12 100644 --- a/docs-website/docs/pipeline-components/preprocessors/documentsplitter.mdx +++ b/docs-website/docs/pipeline-components/preprocessors/documentsplitter.mdx @@ -139,7 +139,7 @@ components: init_parameters: bm25_algorithm: BM25L bm25_parameters: {} - bm25_tokenization_regex: (?u)\\b\\w+\\b + bm25_tokenization_regex: '[^\W\u1100-\u11ff\ua960-\ua97f\ud7b0-\ud7ff\u3130-\u318f\uac00-\ud7af\u3041-\u3096\u30a1-\u30fa\u30fc\u31f0-\u31ff\u3400-\u4dbf\u4e00-\u9fff\uf900-\ufaff\uff66-\uff9f]+|[\u1100-\u11ff\ua960-\ua97f\ud7b0-\ud7ff\u3130-\u318f\uac00-\ud7af\u3041-\u3096\u30a1-\u30fa\u30fc\u31f0-\u31ff\u3400-\u4dbf\u4e00-\u9fff\uf900-\ufaff\uff66-\uff9f]' embedding_similarity_function: dot_product index: 64e4f9ab-87fb-47fd-b390-dabcfda61447 return_embedding: true diff --git a/haystack/document_stores/in_memory/document_store.py b/haystack/document_stores/in_memory/document_store.py index 599ea7d6277..6929bcbc0b7 100644 --- a/haystack/document_stores/in_memory/document_store.py +++ b/haystack/document_stores/in_memory/document_store.py @@ -6,6 +6,7 @@ import json import math import re +import unicodedata import uuid from collections import Counter from collections.abc import Callable, Iterable @@ -35,6 +36,26 @@ BM25_SCALING_FACTOR = 8 DOT_PRODUCT_SCALING_FACTOR = 100 +# Unicode ranges for scripts that are written without spaces between words (Chinese, Japanese, Korean). These +# characters are tokenized one per token so that bare-term queries for individual characters or short words can +# match, while every other script keeps the usual word-based behaviour. The kana ranges intentionally exclude +# punctuation such as the katakana middle dot (・) so it does not become a token on its own. +_CJK_CHAR_CLASS = ( + r"\u1100-\u11ff" # Hangul Jamo + r"\ua960-\ua97f" # Hangul Jamo Extended-A + r"\ud7b0-\ud7ff" # Hangul Jamo Extended-B + r"\u3130-\u318f" # Hangul Compatibility Jamo + r"\uac00-\ud7af" # Hangul Syllables + r"\u3041-\u3096" # Hiragana letters (excludes combining marks and kana iteration marks) + r"\u30a1-\u30fa\u30fc" # Katakana letters + prolonged sound mark, excludes ・ middle dot + r"\u31f0-\u31ff" # Katakana Phonetic Extensions + r"\u3400-\u4dbf" # CJK Unified Ideographs Extension A + r"\u4e00-\u9fff" # CJK Unified Ideographs + r"\uf900-\ufaff" # CJK Compatibility Ideographs + r"\uff66-\uff9f" # Halfwidth Katakana +) +_DEFAULT_BM25_TOKENIZATION_REGEX = rf"[^\W{_CJK_CHAR_CLASS}]+|[{_CJK_CHAR_CLASS}]" + def _make_metadata_value_hashable(value: Any) -> Any: """Convert nested metadata values into values that can be used for deduplication.""" @@ -80,7 +101,7 @@ class InMemoryDocumentStore: def __init__( self, - bm25_tokenization_regex: str = r"(?u)\b\w+\b", + bm25_tokenization_regex: str = _DEFAULT_BM25_TOKENIZATION_REGEX, bm25_algorithm: Literal["BM25Okapi", "BM25L", "BM25Plus"] = "BM25L", bm25_parameters: dict | None = None, embedding_similarity_function: Literal["dot_product", "cosine"] = "dot_product", @@ -94,7 +115,12 @@ def __init__( """ Initializes the DocumentStore. - :param bm25_tokenization_regex: The regular expression used to tokenize the text for BM25 retrieval. + :param bm25_tokenization_regex: + The regular expression used to tokenize the text for BM25 retrieval. The default groups the + word-like characters of space-delimited scripts into word tokens while splitting Chinese, Japanese + and Korean text one character per token, so bare-term queries for single CJK characters or short + words still match; see ``_DEFAULT_BM25_TOKENIZATION_REGEX`` for the exact pattern. Text is Unicode + NFC-normalized before tokenization (see ``_tokenize_bm25``), so equivalent spellings share tokens. :param bm25_algorithm: The BM25 algorithm to use. One of "BM25Okapi", "BM25L", or "BM25Plus". :param bm25_parameters: Parameters for BM25 implementation in a dictionary format. For example: `{'k1':1.5, 'b':0.75, 'epsilon':0.25}` @@ -212,15 +238,15 @@ def _tokenize_bm25(self, text: str) -> list[str]: Here we explicitly create a tokenization method to encapsulate all pre-processing logic used to create BM25 tokens, such as - lowercasing. This helps track the exact tokenization process - used for BM25 scoring at any given time. + lowercasing and Unicode NFC normalization. This helps track the exact tokenization process used for + BM25 scoring at any given time. :param text: The text to tokenize. :returns: A list of tokens. """ - text = text.lower() + text = unicodedata.normalize("NFC", text.lower()) return self.tokenizer(text) def _score_bm25l(self, query: str, documents: list[Document]) -> list[tuple[Document, float]]: diff --git a/releasenotes/notes/bm25-cjk-tokenization-1a2b3c4d5e6f7.yaml b/releasenotes/notes/bm25-cjk-tokenization-1a2b3c4d5e6f7.yaml new file mode 100644 index 00000000000..1bc0c004af3 --- /dev/null +++ b/releasenotes/notes/bm25-cjk-tokenization-1a2b3c4d5e6f7.yaml @@ -0,0 +1,11 @@ +--- +fixes: + - | + Fixed BM25 text tokenization in ``InMemoryDocumentStore`` so that CJK scripts are tokenized per character. + Chinese, Japanese and Korean text -- including Hangul (native jamo, compatibility jamo, syllables and the + jamo extension blocks), CJK ideographs, their Extension A and compatibility blocks, kana, and halfwidth + katakana -- is now split one token per character, so bare-term queries for single characters or short words + can match documents where those characters appear inside a longer unspaced run. Kana punctuation such as + the katakana middle dot is no longer emitted as a standalone token, and text is Unicode NFC-normalized + before tokenization so equivalent spellings (for example decomposed vs. composed Hangul) share the same + terms. diff --git a/test/components/retrievers/test_multi_query_embedding_retriever.py b/test/components/retrievers/test_multi_query_embedding_retriever.py index 31df7890d90..1b6b0e7bdcf 100644 --- a/test/components/retrievers/test_multi_query_embedding_retriever.py +++ b/test/components/retrievers/test_multi_query_embedding_retriever.py @@ -15,6 +15,7 @@ from haystack.components.retrievers import InMemoryEmbeddingRetriever, MultiQueryEmbeddingRetriever from haystack.components.writers import DocumentWriter from haystack.document_stores.in_memory import InMemoryDocumentStore +from haystack.document_stores.in_memory.document_store import _DEFAULT_BM25_TOKENIZATION_REGEX from haystack.document_stores.types import DuplicatePolicy @@ -114,7 +115,7 @@ def test_to_dict(self, in_memory_doc_store): "document_store": { "type": "haystack.document_stores.in_memory.document_store.InMemoryDocumentStore", "init_parameters": { - "bm25_tokenization_regex": "(?u)\\b\\w+\\b", + "bm25_tokenization_regex": _DEFAULT_BM25_TOKENIZATION_REGEX, "bm25_algorithm": "BM25L", "bm25_parameters": {}, "embedding_similarity_function": "dot_product", diff --git a/test/components/retrievers/test_multi_query_text_retriever.py b/test/components/retrievers/test_multi_query_text_retriever.py index d8c02c44f9f..ba477a4ac36 100644 --- a/test/components/retrievers/test_multi_query_text_retriever.py +++ b/test/components/retrievers/test_multi_query_text_retriever.py @@ -13,6 +13,7 @@ from haystack.components.retrievers import InMemoryBM25Retriever, MultiQueryTextRetriever from haystack.components.writers import DocumentWriter from haystack.document_stores.in_memory import InMemoryDocumentStore +from haystack.document_stores.in_memory.document_store import _DEFAULT_BM25_TOKENIZATION_REGEX from haystack.document_stores.types import DuplicatePolicy @@ -83,7 +84,7 @@ def test_to_dict(self, in_memory_doc_store): "document_store": { "type": "haystack.document_stores.in_memory.document_store.InMemoryDocumentStore", "init_parameters": { - "bm25_tokenization_regex": "(?u)\\b\\w+\\b", + "bm25_tokenization_regex": _DEFAULT_BM25_TOKENIZATION_REGEX, "bm25_algorithm": "BM25L", "bm25_parameters": {}, "embedding_similarity_function": "dot_product", diff --git a/test/components/retrievers/test_multi_retriever.py b/test/components/retrievers/test_multi_retriever.py index 5a50c912f9c..78349dea76a 100644 --- a/test/components/retrievers/test_multi_retriever.py +++ b/test/components/retrievers/test_multi_retriever.py @@ -20,6 +20,7 @@ from haystack.components.retrievers.types import TextRetriever from haystack.components.writers import DocumentWriter from haystack.document_stores.in_memory import InMemoryDocumentStore +from haystack.document_stores.in_memory.document_store import _DEFAULT_BM25_TOKENIZATION_REGEX from haystack.document_stores.types import DuplicatePolicy from haystack.utils.experimental import ExperimentalWarning @@ -289,7 +290,7 @@ def test_to_dict(self): "document_store": { "type": "haystack.document_stores.in_memory.document_store.InMemoryDocumentStore", "init_parameters": { - "bm25_tokenization_regex": "(?u)\\b\\w+\\b", + "bm25_tokenization_regex": _DEFAULT_BM25_TOKENIZATION_REGEX, "bm25_algorithm": "BM25L", "bm25_parameters": {}, "embedding_similarity_function": "dot_product", diff --git a/test/components/retrievers/test_sentence_window_retriever.py b/test/components/retrievers/test_sentence_window_retriever.py index 7f5f3f0b3d8..72e221835b5 100644 --- a/test/components/retrievers/test_sentence_window_retriever.py +++ b/test/components/retrievers/test_sentence_window_retriever.py @@ -13,6 +13,7 @@ from haystack.components.retrievers import InMemoryBM25Retriever from haystack.components.retrievers.sentence_window_retriever import SentenceWindowRetriever from haystack.document_stores.in_memory import InMemoryDocumentStore +from haystack.document_stores.in_memory.document_store import _DEFAULT_BM25_TOKENIZATION_REGEX class TestSentenceWindowRetriever: @@ -74,7 +75,7 @@ def test_to_dict(self, in_memory_doc_store): "init_parameters": { "bm25_algorithm": "BM25L", "bm25_parameters": {}, - "bm25_tokenization_regex": "(?u)\\b\\w+\\b", + "bm25_tokenization_regex": _DEFAULT_BM25_TOKENIZATION_REGEX, "embedding_similarity_function": "dot_product", "index": ANY, "shared": True, diff --git a/test/components/retrievers/test_text_embedding_retriever.py b/test/components/retrievers/test_text_embedding_retriever.py index c63ddf6279b..28497cb475c 100644 --- a/test/components/retrievers/test_text_embedding_retriever.py +++ b/test/components/retrievers/test_text_embedding_retriever.py @@ -13,6 +13,7 @@ from haystack.components.retrievers import InMemoryEmbeddingRetriever, TextEmbeddingRetriever from haystack.components.writers import DocumentWriter from haystack.document_stores.in_memory import InMemoryDocumentStore +from haystack.document_stores.in_memory.document_store import _DEFAULT_BM25_TOKENIZATION_REGEX from haystack.document_stores.types import DuplicatePolicy @@ -92,7 +93,7 @@ def test_to_dict(self): "document_store": { "type": "haystack.document_stores.in_memory.document_store.InMemoryDocumentStore", "init_parameters": { - "bm25_tokenization_regex": "(?u)\\b\\w+\\b", + "bm25_tokenization_regex": _DEFAULT_BM25_TOKENIZATION_REGEX, "bm25_algorithm": "BM25L", "bm25_parameters": {}, "embedding_similarity_function": "dot_product", diff --git a/test/document_stores/test_in_memory.py b/test/document_stores/test_in_memory.py index f8f31417666..18393e0ccf2 100644 --- a/test/document_stores/test_in_memory.py +++ b/test/document_stores/test_in_memory.py @@ -7,6 +7,7 @@ import logging import math import tempfile +import unicodedata from typing import Literal, cast from unittest.mock import patch @@ -17,6 +18,7 @@ from haystack.document_stores.errors import DocumentStoreError, DuplicateDocumentError from haystack.document_stores.in_memory import InMemoryDocumentStore from haystack.document_stores.in_memory import document_store as in_memory_module +from haystack.document_stores.in_memory.document_store import _DEFAULT_BM25_TOKENIZATION_REGEX from haystack.testing.document_store import ( CountDocumentsByFilterTest, CountUniqueMetadataByFilterTest, @@ -140,7 +142,7 @@ def test_to_dict(self, in_memory_doc_store): assert data == { "type": "haystack.document_stores.in_memory.document_store.InMemoryDocumentStore", "init_parameters": { - "bm25_tokenization_regex": r"(?u)\b\w+\b", + "bm25_tokenization_regex": _DEFAULT_BM25_TOKENIZATION_REGEX, "bm25_algorithm": "BM25L", "bm25_parameters": {}, "embedding_similarity_function": "dot_product", @@ -817,6 +819,46 @@ def test_bm25_tokenization_includes_single_char_tokens(self, in_memory_doc_store tokens = in_memory_doc_store._tokenize_bm25("Luna is a dog") assert tokens == ["luna", "is", "a", "dog"] + def test_bm25_tokenization_splits_cjk_characters(self, in_memory_doc_store): + # A Hangul noun keeps its attached particle in the default tokenizer, + # which makes a bare-noun query have zero overlap with the document + # (e.g. query "서울" vs document containing only "서울은"). Each CJK + # character (Hangul syllables, CJK unified ideographs, kana) should be + # tokenized on its own so BM25 can match the bare form. + tokens = in_memory_doc_store._tokenize_bm25("서울은 대한민국의 수도") + assert "서울은" not in tokens + assert "서" in tokens and "울" in tokens and "은" in tokens + assert "대한민국의" not in tokens + assert "한" in tokens and "국" in tokens and "의" in tokens + + # Mixed script: Latin words and digits keep their previous behaviour. + tokens2 = in_memory_doc_store._tokenize_bm25("Seoul 2026 시티") + assert tokens2[:2] == ["seoul", "2026"] + assert "시" in tokens2 and "티" in tokens2 + + def test_bm25_tokenization_handles_cjk_details(self, in_memory_doc_store): + tokenize = in_memory_doc_store._tokenize_bm25 + + # Kana punctuation such as the katakana middle dot (\u30fb) must not become a token of its own, + # so a name like john (katakana) splits only on the letters. + tokens = tokenize("\u30b8\u30e7\u30f3\u30fb\u30b9\u30df\u30b9") + assert "\u30fb" not in tokens + assert tokens == ["\u30b8", "\u30e7", "\u30f3", "\u30b9", "\u30df", "\u30b9"] + + # Halfwidth katakana (U+FF66-U+FF9F) and CJK Extension A ideographs split per character too. + assert tokenize("\uff71\uff72\uff73") == ["\uff71", "\uff72", "\uff73"] + assert tokenize("\u3400\u3401\u3402") == ["\u3400", "\u3401", "\u3402"] + # CJK Compatibility Ideographs and the two Hangul Jamo Extended blocks split per character too. + assert tokenize("\ufa0e\ufa0f\ufa11") == ["\ufa0e", "\ufa0f", "\ufa11"] + assert tokenize("\ua960\ud7b0") == ["\ua960", "\ud7b0"] + + # NFC-normalized Hangul and its decomposed NFD spelling tokenize identically, so a query in one + # form matches a document written in the other. + nfc = "\uc11c\uc6b8\uc740" + nfd = unicodedata.normalize("NFD", nfc) + assert nfd != nfc + assert tokenize(nfd) == tokenize(nfc) == ["\uc11c", "\uc6b8", "\uc740"] + def test_bm25_retrieval_with_single_char_query(self, in_memory_doc_store): docs = [ Document(content="C programming language"), diff --git a/test/tools/test_pipeline_tool.py b/test/tools/test_pipeline_tool.py index fc72e606c17..87245b9ff9d 100644 --- a/test/tools/test_pipeline_tool.py +++ b/test/tools/test_pipeline_tool.py @@ -15,6 +15,7 @@ from haystack.components.retrievers import InMemoryBM25Retriever, InMemoryEmbeddingRetriever from haystack.dataclasses import ChatMessage from haystack.document_stores.in_memory import InMemoryDocumentStore +from haystack.document_stores.in_memory.document_store import _DEFAULT_BM25_TOKENIZATION_REGEX from haystack.tools import PipelineTool @@ -68,7 +69,7 @@ def sample_pipeline_dict(): "document_store": { "type": "haystack.document_stores.in_memory.document_store.InMemoryDocumentStore", "init_parameters": { - "bm25_tokenization_regex": "(?u)\\b\\w+\\b", + "bm25_tokenization_regex": _DEFAULT_BM25_TOKENIZATION_REGEX, "bm25_algorithm": "BM25L", "bm25_parameters": {}, "embedding_similarity_function": "dot_product", From 7b9f78520a14f31d294cc5ec6c119e279de25920 Mon Sep 17 00:00:00 2001 From: rain <1504569896@qq.com> Date: Sat, 26 Sep 2026 13:37:31 +0800 Subject: [PATCH 2/2] fix: address review feedback for BM25 CJK tokenization - Add halfwidth Hangul Jamo (U+FFA0-U+FFDC) to the CJK char class - Shorten the release note and drop the middle-dot claim (it was never a token on main) - Stop referencing private names in the public docstring, use single backticks - Assert full token lists in the CJK tokenization test - Add a retrieval test for the bare-term CJK query the PR fixes --- .../in_memory/document_store.py | 9 +++---- .../bm25-cjk-tokenization-1a2b3c4d5e6f7.yaml | 12 +++------ test/document_stores/test_in_memory.py | 26 ++++++++++++++----- 3 files changed, 27 insertions(+), 20 deletions(-) diff --git a/haystack/document_stores/in_memory/document_store.py b/haystack/document_stores/in_memory/document_store.py index 6929bcbc0b7..27278e5fd94 100644 --- a/haystack/document_stores/in_memory/document_store.py +++ b/haystack/document_stores/in_memory/document_store.py @@ -53,6 +53,7 @@ r"\u4e00-\u9fff" # CJK Unified Ideographs r"\uf900-\ufaff" # CJK Compatibility Ideographs r"\uff66-\uff9f" # Halfwidth Katakana + r"\uffa0-\uffdc" # Halfwidth Hangul Jamo ) _DEFAULT_BM25_TOKENIZATION_REGEX = rf"[^\W{_CJK_CHAR_CLASS}]+|[{_CJK_CHAR_CLASS}]" @@ -116,11 +117,9 @@ def __init__( Initializes the DocumentStore. :param bm25_tokenization_regex: - The regular expression used to tokenize the text for BM25 retrieval. The default groups the - word-like characters of space-delimited scripts into word tokens while splitting Chinese, Japanese - and Korean text one character per token, so bare-term queries for single CJK characters or short - words still match; see ``_DEFAULT_BM25_TOKENIZATION_REGEX`` for the exact pattern. Text is Unicode - NFC-normalized before tokenization (see ``_tokenize_bm25``), so equivalent spellings share tokens. + The regular expression used to tokenize the text for BM25 retrieval. The default groups word + characters into word tokens and splits Chinese, Japanese and Korean text into one token per character. + Text is lowercased and NFC-normalized before tokenization. :param bm25_algorithm: The BM25 algorithm to use. One of "BM25Okapi", "BM25L", or "BM25Plus". :param bm25_parameters: Parameters for BM25 implementation in a dictionary format. For example: `{'k1':1.5, 'b':0.75, 'epsilon':0.25}` diff --git a/releasenotes/notes/bm25-cjk-tokenization-1a2b3c4d5e6f7.yaml b/releasenotes/notes/bm25-cjk-tokenization-1a2b3c4d5e6f7.yaml index 1bc0c004af3..5bdec7628c6 100644 --- a/releasenotes/notes/bm25-cjk-tokenization-1a2b3c4d5e6f7.yaml +++ b/releasenotes/notes/bm25-cjk-tokenization-1a2b3c4d5e6f7.yaml @@ -1,11 +1,7 @@ --- fixes: - | - Fixed BM25 text tokenization in ``InMemoryDocumentStore`` so that CJK scripts are tokenized per character. - Chinese, Japanese and Korean text -- including Hangul (native jamo, compatibility jamo, syllables and the - jamo extension blocks), CJK ideographs, their Extension A and compatibility blocks, kana, and halfwidth - katakana -- is now split one token per character, so bare-term queries for single characters or short words - can match documents where those characters appear inside a longer unspaced run. Kana punctuation such as - the katakana middle dot is no longer emitted as a standalone token, and text is Unicode NFC-normalized - before tokenization so equivalent spellings (for example decomposed vs. composed Hangul) share the same - terms. + Fixed BM25 tokenization in ``InMemoryDocumentStore`` for Chinese, Japanese and Korean text. The default + ``bm25_tokenization_regex`` now splits CJK characters into one token each, so bare-term queries can match + words inside longer unspaced runs. Text is also NFC-normalized before tokenization, so composed and + decomposed spellings produce the same tokens. diff --git a/test/document_stores/test_in_memory.py b/test/document_stores/test_in_memory.py index 18393e0ccf2..2496b57bfd2 100644 --- a/test/document_stores/test_in_memory.py +++ b/test/document_stores/test_in_memory.py @@ -826,15 +826,11 @@ def test_bm25_tokenization_splits_cjk_characters(self, in_memory_doc_store): # character (Hangul syllables, CJK unified ideographs, kana) should be # tokenized on its own so BM25 can match the bare form. tokens = in_memory_doc_store._tokenize_bm25("서울은 대한민국의 수도") - assert "서울은" not in tokens - assert "서" in tokens and "울" in tokens and "은" in tokens - assert "대한민국의" not in tokens - assert "한" in tokens and "국" in tokens and "의" in tokens + assert tokens == ["서", "울", "은", "대", "한", "민", "국", "의", "수", "도"] # Mixed script: Latin words and digits keep their previous behaviour. - tokens2 = in_memory_doc_store._tokenize_bm25("Seoul 2026 시티") - assert tokens2[:2] == ["seoul", "2026"] - assert "시" in tokens2 and "티" in tokens2 + tokens = in_memory_doc_store._tokenize_bm25("Seoul 2026 시티") + assert tokens == ["seoul", "2026", "시", "티"] def test_bm25_tokenization_handles_cjk_details(self, in_memory_doc_store): tokenize = in_memory_doc_store._tokenize_bm25 @@ -851,6 +847,8 @@ def test_bm25_tokenization_handles_cjk_details(self, in_memory_doc_store): # CJK Compatibility Ideographs and the two Hangul Jamo Extended blocks split per character too. assert tokenize("\ufa0e\ufa0f\ufa11") == ["\ufa0e", "\ufa0f", "\ufa11"] assert tokenize("\ua960\ud7b0") == ["\ua960", "\ud7b0"] + # Halfwidth Hangul Jamo (U+FFA0-U+FFDC), the Korean counterpart of halfwidth katakana, too. + assert tokenize("\uffb1\uffb2\uffb3") == ["\uffb1", "\uffb2", "\uffb3"] # NFC-normalized Hangul and its decomposed NFD spelling tokenize identically, so a query in one # form matches a document written in the other. @@ -871,6 +869,20 @@ def test_bm25_retrieval_with_single_char_query(self, in_memory_doc_store): assert len(results) == 1 assert results[0].content == "C programming language" + def test_bm25_retrieval_with_cjk_bare_term_query(self, in_memory_doc_store): + # The bug this PR fixes: a document holds the noun with its attached + # particle ("서울은"), so the default tokenizer had zero overlap with the + # bare query "서울" and retrieval silently returned nothing. + docs = [ + Document(content="서울은 대한민국의 수도이며 인구가 가장 많다."), + Document(content="부산은 대한민국 제2의 도시이자 최대 항구이다."), + ] + in_memory_doc_store.write_documents(docs) + + results = in_memory_doc_store.bm25_retrieval(query="서울", top_k=1) + assert len(results) == 1 + assert results[0].content == "서울은 대한민국의 수도이며 인구가 가장 많다." + def test_bm25_retrieval_single_char_content_token(self, in_memory_doc_store): docs = [Document(content="I like R"), Document(content="I like Python")] in_memory_doc_store.write_documents(docs)