Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion haystack/token_counters/tiktoken_counter.py
Original file line number Diff line number Diff line change
Expand Up @@ -74,7 +74,7 @@ def count(self, messages: list[ChatMessage], tools: ToolsType | None = None) ->
if not messages and not tools:
return 0
self.warm_up()
text_tokens = len(self._encoder.encode(_rendered_conversation(messages) + _rendered_tools(tools)))
text_tokens = len(self._encoder.encode_ordinary(_rendered_conversation(messages) + _rendered_tools(tools)))
return text_tokens + _non_text_tokens(
messages, tokens_per_image=self.tokens_per_image, tokens_per_file=self.tokens_per_file
)
Expand Down
21 changes: 21 additions & 0 deletions test/token_counters/test_tiktoken_counter.py
Original file line number Diff line number Diff line change
Expand Up @@ -31,6 +31,10 @@ def encode(self, text: str) -> list[int]:
self.encoded.append(text)
return list(range(len(text.split())))

def encode_ordinary(self, text: str) -> list[int]:
self.encoded.append(text)
return list(range(len(text.split())))


@pytest.fixture
def fake_encoder(monkeypatch: pytest.MonkeyPatch) -> _FakeEncoder:
Expand Down Expand Up @@ -154,6 +158,23 @@ def test_every_message_contributes(self):

assert counter.count(messages) > max(counter.count([message]) for message in messages)

def test_literal_special_token_strings_in_messages_do_not_raise(self):
# Compaction may count tool results or retrieved docs that quote tiktoken markers.
counter = TiktokenCounter()
count = counter.count(
[ChatMessage.from_user("The manual documents <|endoftext|> as a literal marker.")]
)
assert count > 0

def test_literal_special_token_strings_in_tool_descriptions_do_not_raise(self):
@tool
def lookup(query: Annotated[str, "the query"]) -> str:
"""Looks up documentation mentioning <|endoftext|> as a literal marker."""
return "ok"

count = TiktokenCounter().count([], tools=[lookup])
assert count > 0

def test_an_image_is_charged_at_the_flat_rate(self):
# A tokenizer cannot price an image, so it gets a flat estimate rather than the handful of tokens its
# placeholder text would cost.
Expand Down