From 1d8bbcc3eb66d0ad75eda8fbf2a2d85f4217732a Mon Sep 17 00:00:00 2001
From: sjrl <10526848+sjrl@users.noreply.github.com>
Date: Mon, 28 Sep 2026 06:38:54 +0000
Subject: [PATCH] Sync Haystack API reference on Docusaurus
---
docs-website/reference/haystack-api/document_stores_api.md | 6 ++++--
1 file changed, 4 insertions(+), 2 deletions(-)
diff --git a/docs-website/reference/haystack-api/document_stores_api.md b/docs-website/reference/haystack-api/document_stores_api.md
index ced0521d0a2..395dc64a2e4 100644
--- a/docs-website/reference/haystack-api/document_stores_api.md
+++ b/docs-website/reference/haystack-api/document_stores_api.md
@@ -25,7 +25,7 @@ Stores data in-memory. It's ephemeral and cannot be saved to disk.
```python
__init__(
- bm25_tokenization_regex: str = "(?u)\\b\\w+\\b",
+ bm25_tokenization_regex: str = _DEFAULT_BM25_TOKENIZATION_REGEX,
bm25_algorithm: Literal["BM25Okapi", "BM25L", "BM25Plus"] = "BM25L",
bm25_parameters: dict | None = None,
embedding_similarity_function: Literal[
@@ -44,7 +44,9 @@ Initializes the DocumentStore.
**Parameters:**
-- **bm25_tokenization_regex** (str) – The regular expression used to tokenize the text for BM25 retrieval.
+- **bm25_tokenization_regex** (str) – The regular expression used to tokenize the text for BM25 retrieval. The default groups word
+ characters into word tokens and splits Chinese, Japanese and Korean text into one token per character.
+ Text is lowercased and NFC-normalized before tokenization.
- **bm25_algorithm** (Literal['BM25Okapi', 'BM25L', 'BM25Plus']) – The BM25 algorithm to use. One of "BM25Okapi", "BM25L", or "BM25Plus".
- **bm25_parameters** (dict | None) – Parameters for BM25 implementation in a dictionary format.
For example: `{'k1':1.5, 'b':0.75, 'epsilon':0.25}`