Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -137,7 +137,7 @@ components:
init_parameters:
bm25_algorithm: BM25L
bm25_parameters: {}
bm25_tokenization_regex: (?u)\\b\\w+\\b
bm25_tokenization_regex: '[^\W\u1100-\u11ff\ua960-\ua97f\ud7b0-\ud7ff\u3130-\u318f\uac00-\ud7af\u3041-\u3096\u30a1-\u30fa\u30fc\u31f0-\u31ff\u3400-\u4dbf\u4e00-\u9fff\uf900-\ufaff\uff66-\uff9f]+|[\u1100-\u11ff\ua960-\ua97f\ud7b0-\ud7ff\u3130-\u318f\uac00-\ud7af\u3041-\u3096\u30a1-\u30fa\u30fc\u31f0-\u31ff\u3400-\u4dbf\u4e00-\u9fff\uf900-\ufaff\uff66-\uff9f]'
embedding_similarity_function: dot_product
index: 64e4f9ab-87fb-47fd-b390-dabcfda61447
return_embedding: true
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -139,7 +139,7 @@ components:
init_parameters:
bm25_algorithm: BM25L
bm25_parameters: {}
bm25_tokenization_regex: (?u)\\b\\w+\\b
bm25_tokenization_regex: '[^\W\u1100-\u11ff\ua960-\ua97f\ud7b0-\ud7ff\u3130-\u318f\uac00-\ud7af\u3041-\u3096\u30a1-\u30fa\u30fc\u31f0-\u31ff\u3400-\u4dbf\u4e00-\u9fff\uf900-\ufaff\uff66-\uff9f]+|[\u1100-\u11ff\ua960-\ua97f\ud7b0-\ud7ff\u3130-\u318f\uac00-\ud7af\u3041-\u3096\u30a1-\u30fa\u30fc\u31f0-\u31ff\u3400-\u4dbf\u4e00-\u9fff\uf900-\ufaff\uff66-\uff9f]'
embedding_similarity_function: dot_product
index: 64e4f9ab-87fb-47fd-b390-dabcfda61447
return_embedding: true
Expand Down
35 changes: 30 additions & 5 deletions haystack/document_stores/in_memory/document_store.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,7 @@
import json
import math
import re
import unicodedata
import uuid
from collections import Counter
from collections.abc import Callable, Iterable
Expand Down Expand Up @@ -35,6 +36,27 @@
BM25_SCALING_FACTOR = 8
DOT_PRODUCT_SCALING_FACTOR = 100

# Unicode ranges for scripts that are written without spaces between words (Chinese, Japanese, Korean). These
# characters are tokenized one per token so that bare-term queries for individual characters or short words can
# match, while every other script keeps the usual word-based behaviour. The kana ranges intentionally exclude
# punctuation such as the katakana middle dot (・) so it does not become a token on its own.
_CJK_CHAR_CLASS = (
r"\u1100-\u11ff" # Hangul Jamo
r"\ua960-\ua97f" # Hangul Jamo Extended-A
r"\ud7b0-\ud7ff" # Hangul Jamo Extended-B
r"\u3130-\u318f" # Hangul Compatibility Jamo
r"\uac00-\ud7af" # Hangul Syllables
r"\u3041-\u3096" # Hiragana letters (excludes combining marks and kana iteration marks)
r"\u30a1-\u30fa\u30fc" # Katakana letters + prolonged sound mark, excludes ・ middle dot
r"\u31f0-\u31ff" # Katakana Phonetic Extensions
r"\u3400-\u4dbf" # CJK Unified Ideographs Extension A
r"\u4e00-\u9fff" # CJK Unified Ideographs
r"\uf900-\ufaff" # CJK Compatibility Ideographs
r"\uff66-\uff9f" # Halfwidth Katakana
r"\uffa0-\uffdc" # Halfwidth Hangul Jamo
)
_DEFAULT_BM25_TOKENIZATION_REGEX = rf"[^\W{_CJK_CHAR_CLASS}]+|[{_CJK_CHAR_CLASS}]"


def _make_metadata_value_hashable(value: Any) -> Any:
"""Convert nested metadata values into values that can be used for deduplication."""
Expand Down Expand Up @@ -80,7 +102,7 @@ class InMemoryDocumentStore:

def __init__(
self,
bm25_tokenization_regex: str = r"(?u)\b\w+\b",
bm25_tokenization_regex: str = _DEFAULT_BM25_TOKENIZATION_REGEX,
bm25_algorithm: Literal["BM25Okapi", "BM25L", "BM25Plus"] = "BM25L",
bm25_parameters: dict | None = None,
embedding_similarity_function: Literal["dot_product", "cosine"] = "dot_product",
Expand All @@ -94,7 +116,10 @@ def __init__(
"""
Initializes the DocumentStore.

:param bm25_tokenization_regex: The regular expression used to tokenize the text for BM25 retrieval.
:param bm25_tokenization_regex:
The regular expression used to tokenize the text for BM25 retrieval. The default groups word
characters into word tokens and splits Chinese, Japanese and Korean text into one token per character.
Text is lowercased and NFC-normalized before tokenization.
:param bm25_algorithm: The BM25 algorithm to use. One of "BM25Okapi", "BM25L", or "BM25Plus".
:param bm25_parameters: Parameters for BM25 implementation in a dictionary format.
For example: `{'k1':1.5, 'b':0.75, 'epsilon':0.25}`
Expand Down Expand Up @@ -212,15 +237,15 @@ def _tokenize_bm25(self, text: str) -> list[str]:

Here we explicitly create a tokenization method to encapsulate
all pre-processing logic used to create BM25 tokens, such as
lowercasing. This helps track the exact tokenization process
used for BM25 scoring at any given time.
lowercasing and Unicode NFC normalization. This helps track the exact tokenization process used for
BM25 scoring at any given time.

:param text:
The text to tokenize.
:returns:
A list of tokens.
"""
text = text.lower()
text = unicodedata.normalize("NFC", text.lower())
return self.tokenizer(text)

def _score_bm25l(self, query: str, documents: list[Document]) -> list[tuple[Document, float]]:
Expand Down
7 changes: 7 additions & 0 deletions releasenotes/notes/bm25-cjk-tokenization-1a2b3c4d5e6f7.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,7 @@
---
fixes:
- |
Fixed BM25 tokenization in ``InMemoryDocumentStore`` for Chinese, Japanese and Korean text. The default
``bm25_tokenization_regex`` now splits CJK characters into one token each, so bare-term queries can match
words inside longer unspaced runs. Text is also NFC-normalized before tokenization, so composed and
decomposed spellings produce the same tokens.
Original file line number Diff line number Diff line change
Expand Up @@ -15,6 +15,7 @@
from haystack.components.retrievers import InMemoryEmbeddingRetriever, MultiQueryEmbeddingRetriever
from haystack.components.writers import DocumentWriter
from haystack.document_stores.in_memory import InMemoryDocumentStore
from haystack.document_stores.in_memory.document_store import _DEFAULT_BM25_TOKENIZATION_REGEX
from haystack.document_stores.types import DuplicatePolicy


Expand Down Expand Up @@ -114,7 +115,7 @@ def test_to_dict(self, in_memory_doc_store):
"document_store": {
"type": "haystack.document_stores.in_memory.document_store.InMemoryDocumentStore",
"init_parameters": {
"bm25_tokenization_regex": "(?u)\\b\\w+\\b",
"bm25_tokenization_regex": _DEFAULT_BM25_TOKENIZATION_REGEX,
"bm25_algorithm": "BM25L",
"bm25_parameters": {},
"embedding_similarity_function": "dot_product",
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -13,6 +13,7 @@
from haystack.components.retrievers import InMemoryBM25Retriever, MultiQueryTextRetriever
from haystack.components.writers import DocumentWriter
from haystack.document_stores.in_memory import InMemoryDocumentStore
from haystack.document_stores.in_memory.document_store import _DEFAULT_BM25_TOKENIZATION_REGEX
from haystack.document_stores.types import DuplicatePolicy


Expand Down Expand Up @@ -83,7 +84,7 @@ def test_to_dict(self, in_memory_doc_store):
"document_store": {
"type": "haystack.document_stores.in_memory.document_store.InMemoryDocumentStore",
"init_parameters": {
"bm25_tokenization_regex": "(?u)\\b\\w+\\b",
"bm25_tokenization_regex": _DEFAULT_BM25_TOKENIZATION_REGEX,
"bm25_algorithm": "BM25L",
"bm25_parameters": {},
"embedding_similarity_function": "dot_product",
Expand Down
3 changes: 2 additions & 1 deletion test/components/retrievers/test_multi_retriever.py
Original file line number Diff line number Diff line change
Expand Up @@ -20,6 +20,7 @@
from haystack.components.retrievers.types import TextRetriever
from haystack.components.writers import DocumentWriter
from haystack.document_stores.in_memory import InMemoryDocumentStore
from haystack.document_stores.in_memory.document_store import _DEFAULT_BM25_TOKENIZATION_REGEX
from haystack.document_stores.types import DuplicatePolicy
from haystack.utils.experimental import ExperimentalWarning

Expand Down Expand Up @@ -289,7 +290,7 @@ def test_to_dict(self):
"document_store": {
"type": "haystack.document_stores.in_memory.document_store.InMemoryDocumentStore",
"init_parameters": {
"bm25_tokenization_regex": "(?u)\\b\\w+\\b",
"bm25_tokenization_regex": _DEFAULT_BM25_TOKENIZATION_REGEX,
"bm25_algorithm": "BM25L",
"bm25_parameters": {},
"embedding_similarity_function": "dot_product",
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -13,6 +13,7 @@
from haystack.components.retrievers import InMemoryBM25Retriever
from haystack.components.retrievers.sentence_window_retriever import SentenceWindowRetriever
from haystack.document_stores.in_memory import InMemoryDocumentStore
from haystack.document_stores.in_memory.document_store import _DEFAULT_BM25_TOKENIZATION_REGEX


class TestSentenceWindowRetriever:
Expand Down Expand Up @@ -74,7 +75,7 @@ def test_to_dict(self, in_memory_doc_store):
"init_parameters": {
"bm25_algorithm": "BM25L",
"bm25_parameters": {},
"bm25_tokenization_regex": "(?u)\\b\\w+\\b",
"bm25_tokenization_regex": _DEFAULT_BM25_TOKENIZATION_REGEX,
"embedding_similarity_function": "dot_product",
"index": ANY,
"shared": True,
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -13,6 +13,7 @@
from haystack.components.retrievers import InMemoryEmbeddingRetriever, TextEmbeddingRetriever
from haystack.components.writers import DocumentWriter
from haystack.document_stores.in_memory import InMemoryDocumentStore
from haystack.document_stores.in_memory.document_store import _DEFAULT_BM25_TOKENIZATION_REGEX
from haystack.document_stores.types import DuplicatePolicy


Expand Down Expand Up @@ -92,7 +93,7 @@ def test_to_dict(self):
"document_store": {
"type": "haystack.document_stores.in_memory.document_store.InMemoryDocumentStore",
"init_parameters": {
"bm25_tokenization_regex": "(?u)\\b\\w+\\b",
"bm25_tokenization_regex": _DEFAULT_BM25_TOKENIZATION_REGEX,
"bm25_algorithm": "BM25L",
"bm25_parameters": {},
"embedding_similarity_function": "dot_product",
Expand Down
56 changes: 55 additions & 1 deletion test/document_stores/test_in_memory.py
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,7 @@
import logging
import math
import tempfile
import unicodedata
from typing import Literal, cast
from unittest.mock import patch

Expand All @@ -17,6 +18,7 @@
from haystack.document_stores.errors import DocumentStoreError, DuplicateDocumentError
from haystack.document_stores.in_memory import InMemoryDocumentStore
from haystack.document_stores.in_memory import document_store as in_memory_module
from haystack.document_stores.in_memory.document_store import _DEFAULT_BM25_TOKENIZATION_REGEX
from haystack.testing.document_store import (
CountDocumentsByFilterTest,
CountUniqueMetadataByFilterTest,
Expand Down Expand Up @@ -140,7 +142,7 @@ def test_to_dict(self, in_memory_doc_store):
assert data == {
"type": "haystack.document_stores.in_memory.document_store.InMemoryDocumentStore",
"init_parameters": {
"bm25_tokenization_regex": r"(?u)\b\w+\b",
"bm25_tokenization_regex": _DEFAULT_BM25_TOKENIZATION_REGEX,
"bm25_algorithm": "BM25L",
"bm25_parameters": {},
"embedding_similarity_function": "dot_product",
Expand Down Expand Up @@ -817,6 +819,44 @@ def test_bm25_tokenization_includes_single_char_tokens(self, in_memory_doc_store
tokens = in_memory_doc_store._tokenize_bm25("Luna is a dog")
assert tokens == ["luna", "is", "a", "dog"]

def test_bm25_tokenization_splits_cjk_characters(self, in_memory_doc_store):
Comment thread
sjrl marked this conversation as resolved.
# A Hangul noun keeps its attached particle in the default tokenizer,
# which makes a bare-noun query have zero overlap with the document
# (e.g. query "서울" vs document containing only "서울은"). Each CJK
# character (Hangul syllables, CJK unified ideographs, kana) should be
# tokenized on its own so BM25 can match the bare form.
tokens = in_memory_doc_store._tokenize_bm25("서울은 대한민국의 수도")
assert tokens == ["서", "울", "은", "대", "한", "민", "국", "의", "수", "도"]

# Mixed script: Latin words and digits keep their previous behaviour.
tokens = in_memory_doc_store._tokenize_bm25("Seoul 2026 시티")
assert tokens == ["seoul", "2026", "시", "티"]

def test_bm25_tokenization_handles_cjk_details(self, in_memory_doc_store):
tokenize = in_memory_doc_store._tokenize_bm25

# Kana punctuation such as the katakana middle dot (\u30fb) must not become a token of its own,
# so a name like john (katakana) splits only on the letters.
tokens = tokenize("\u30b8\u30e7\u30f3\u30fb\u30b9\u30df\u30b9")
assert "\u30fb" not in tokens
assert tokens == ["\u30b8", "\u30e7", "\u30f3", "\u30b9", "\u30df", "\u30b9"]

# Halfwidth katakana (U+FF66-U+FF9F) and CJK Extension A ideographs split per character too.
assert tokenize("\uff71\uff72\uff73") == ["\uff71", "\uff72", "\uff73"]
assert tokenize("\u3400\u3401\u3402") == ["\u3400", "\u3401", "\u3402"]
# CJK Compatibility Ideographs and the two Hangul Jamo Extended blocks split per character too.
assert tokenize("\ufa0e\ufa0f\ufa11") == ["\ufa0e", "\ufa0f", "\ufa11"]
assert tokenize("\ua960\ud7b0") == ["\ua960", "\ud7b0"]
# Halfwidth Hangul Jamo (U+FFA0-U+FFDC), the Korean counterpart of halfwidth katakana, too.
assert tokenize("\uffb1\uffb2\uffb3") == ["\uffb1", "\uffb2", "\uffb3"]

# NFC-normalized Hangul and its decomposed NFD spelling tokenize identically, so a query in one
# form matches a document written in the other.
nfc = "\uc11c\uc6b8\uc740"
nfd = unicodedata.normalize("NFD", nfc)
assert nfd != nfc
assert tokenize(nfd) == tokenize(nfc) == ["\uc11c", "\uc6b8", "\uc740"]

def test_bm25_retrieval_with_single_char_query(self, in_memory_doc_store):
docs = [
Document(content="C programming language"),
Expand All @@ -829,6 +869,20 @@ def test_bm25_retrieval_with_single_char_query(self, in_memory_doc_store):
assert len(results) == 1
assert results[0].content == "C programming language"

def test_bm25_retrieval_with_cjk_bare_term_query(self, in_memory_doc_store):
# The bug this PR fixes: a document holds the noun with its attached
# particle ("서울은"), so the default tokenizer had zero overlap with the
# bare query "서울" and retrieval silently returned nothing.
docs = [
Document(content="서울은 대한민국의 수도이며 인구가 가장 많다."),
Document(content="부산은 대한민국 제2의 도시이자 최대 항구이다."),
]
in_memory_doc_store.write_documents(docs)

results = in_memory_doc_store.bm25_retrieval(query="서울", top_k=1)
assert len(results) == 1
assert results[0].content == "서울은 대한민국의 수도이며 인구가 가장 많다."

def test_bm25_retrieval_single_char_content_token(self, in_memory_doc_store):
docs = [Document(content="I like R"), Document(content="I like Python")]
in_memory_doc_store.write_documents(docs)
Expand Down
3 changes: 2 additions & 1 deletion test/tools/test_pipeline_tool.py
Original file line number Diff line number Diff line change
Expand Up @@ -15,6 +15,7 @@
from haystack.components.retrievers import InMemoryBM25Retriever, InMemoryEmbeddingRetriever
from haystack.dataclasses import ChatMessage
from haystack.document_stores.in_memory import InMemoryDocumentStore
from haystack.document_stores.in_memory.document_store import _DEFAULT_BM25_TOKENIZATION_REGEX
from haystack.tools import PipelineTool


Expand Down Expand Up @@ -68,7 +69,7 @@ def sample_pipeline_dict():
"document_store": {
"type": "haystack.document_stores.in_memory.document_store.InMemoryDocumentStore",
"init_parameters": {
"bm25_tokenization_regex": "(?u)\\b\\w+\\b",
"bm25_tokenization_regex": _DEFAULT_BM25_TOKENIZATION_REGEX,
"bm25_algorithm": "BM25L",
"bm25_parameters": {},
"embedding_similarity_function": "dot_product",
Expand Down
Loading