diff --git a/haystack/token_counters/tiktoken_counter.py b/haystack/token_counters/tiktoken_counter.py index 57298234074..f505b89bf99 100644 --- a/haystack/token_counters/tiktoken_counter.py +++ b/haystack/token_counters/tiktoken_counter.py @@ -74,7 +74,7 @@ def count(self, messages: list[ChatMessage], tools: ToolsType | None = None) -> if not messages and not tools: return 0 self.warm_up() - text_tokens = len(self._encoder.encode(_rendered_conversation(messages) + _rendered_tools(tools))) + text_tokens = len(self._encoder.encode_ordinary(_rendered_conversation(messages) + _rendered_tools(tools))) return text_tokens + _non_text_tokens( messages, tokens_per_image=self.tokens_per_image, tokens_per_file=self.tokens_per_file ) diff --git a/releasenotes/notes/Fix-TiktokenCounter-crash-on-literal-special-token-strings-9c0c75f8a3e03c50.yaml b/releasenotes/notes/Fix-TiktokenCounter-crash-on-literal-special-token-strings-9c0c75f8a3e03c50.yaml new file mode 100644 index 00000000000..b7aa0412f49 --- /dev/null +++ b/releasenotes/notes/Fix-TiktokenCounter-crash-on-literal-special-token-strings-9c0c75f8a3e03c50.yaml @@ -0,0 +1,8 @@ +--- +fixes: + - | + Fix ``TiktokenCounter`` crashing with ``ValueError`` when messages or + tool schemas contain literal special-token strings such as + ``<|endoftext|>``. Counting now uses ``encode_ordinary``, matching the + earlier fix for the document splitters in `#12853 + `_. diff --git a/test/token_counters/test_tiktoken_counter.py b/test/token_counters/test_tiktoken_counter.py index bd19f1e5871..bc78f6d8083 100644 --- a/test/token_counters/test_tiktoken_counter.py +++ b/test/token_counters/test_tiktoken_counter.py @@ -27,7 +27,10 @@ class _FakeEncoder: def __init__(self) -> None: self.encoded: list[str] = [] - def encode(self, text: str) -> list[int]: + def encode_ordinary(self, text: str) -> list[int]: + # Mirrors `tiktoken.Encoding.encode_ordinary`: literal special-token + # markers (e.g. `<|endoftext|>`) are counted as ordinary text and + # must never raise. See https://github.com/deepset-ai/haystack/issues/12869 self.encoded.append(text) return list(range(len(text.split()))) @@ -131,6 +134,13 @@ def test_nothing_to_measure_is_zero(self): assert TiktokenCounter().count([]) == 0 assert TiktokenCounter().count([], tools=None) == 0 + def test_special_token_markers_are_counted_as_ordinary_text(self, fake_encoder): + # https://github.com/deepset-ai/haystack/issues/12869 — `encode()` + # raises on literal `<|endoftext|>`; `encode_ordinary()` must not. + messages = [ChatMessage.from_user("score <|endoftext|> now")] + assert TiktokenCounter().count(messages) > 0 + assert "<|endoftext|>" in fake_encoder.encoded[0] + @pytest.mark.integration class TestTiktokenCounterIntegration: