From 37f10ea91b9ec0db8b4ae227488cf64b734ed8c6 Mon Sep 17 00:00:00 2001 From: Harsh Raj Singhania Date: Wed, 23 Sep 2026 16:06:16 +0530 Subject: [PATCH] fix: count literal tiktoken special tokens as ordinary text TiktokenCounter.count() used Encoding.encode(), which raises ValueError when user text contains markers such as <|endoftext|>. Switch to encode_ordinary() so those strings are counted as regular text, matching DocumentSplitter after #12853. Fixes #12869 --- haystack/token_counters/tiktoken_counter.py | 2 +- test/token_counters/test_tiktoken_counter.py | 21 ++++++++++++++++++++ 2 files changed, 22 insertions(+), 1 deletion(-) diff --git a/haystack/token_counters/tiktoken_counter.py b/haystack/token_counters/tiktoken_counter.py index 57298234074..f505b89bf99 100644 --- a/haystack/token_counters/tiktoken_counter.py +++ b/haystack/token_counters/tiktoken_counter.py @@ -74,7 +74,7 @@ def count(self, messages: list[ChatMessage], tools: ToolsType | None = None) -> if not messages and not tools: return 0 self.warm_up() - text_tokens = len(self._encoder.encode(_rendered_conversation(messages) + _rendered_tools(tools))) + text_tokens = len(self._encoder.encode_ordinary(_rendered_conversation(messages) + _rendered_tools(tools))) return text_tokens + _non_text_tokens( messages, tokens_per_image=self.tokens_per_image, tokens_per_file=self.tokens_per_file ) diff --git a/test/token_counters/test_tiktoken_counter.py b/test/token_counters/test_tiktoken_counter.py index bd19f1e5871..5c23c36ad65 100644 --- a/test/token_counters/test_tiktoken_counter.py +++ b/test/token_counters/test_tiktoken_counter.py @@ -31,6 +31,10 @@ def encode(self, text: str) -> list[int]: self.encoded.append(text) return list(range(len(text.split()))) + def encode_ordinary(self, text: str) -> list[int]: + self.encoded.append(text) + return list(range(len(text.split()))) + @pytest.fixture def fake_encoder(monkeypatch: pytest.MonkeyPatch) -> _FakeEncoder: @@ -154,6 +158,23 @@ def test_every_message_contributes(self): assert counter.count(messages) > max(counter.count([message]) for message in messages) + def test_literal_special_token_strings_in_messages_do_not_raise(self): + # Compaction may count tool results or retrieved docs that quote tiktoken markers. + counter = TiktokenCounter() + count = counter.count( + [ChatMessage.from_user("The manual documents <|endoftext|> as a literal marker.")] + ) + assert count > 0 + + def test_literal_special_token_strings_in_tool_descriptions_do_not_raise(self): + @tool + def lookup(query: Annotated[str, "the query"]) -> str: + """Looks up documentation mentioning <|endoftext|> as a literal marker.""" + return "ok" + + count = TiktokenCounter().count([], tools=[lookup]) + assert count > 0 + def test_an_image_is_charged_at_the_flat_rate(self): # A tokenizer cannot price an image, so it gets a flat estimate rather than the handful of tokens its # placeholder text would cost.