Spaces:
Running on Zero
Running on Zero
| import pytest | |
| def test_tokenize_returns_list_of_strings(loaded_models): | |
| from llm_backend.tokenizer import tokenize | |
| tokens = tokenize("The cat sat on the mat") | |
| assert isinstance(tokens, list) | |
| assert len(tokens) > 0 | |
| for t in tokens: | |
| assert isinstance(t, str) | |
| def test_tokenize_empty_returns_empty(loaded_models): | |
| from llm_backend.tokenizer import tokenize | |
| assert tokenize("") == [] | |
| assert tokenize(" ") == [] | |
| def test_tokenize_produces_subword_tokens(loaded_models): | |
| from llm_backend.tokenizer import tokenize | |
| tokens = tokenize("unbelievable") | |
| assert len(tokens) >= 1 | |
| full = "".join(tokens) | |
| assert full == "unbelievable" | |
| def test_token_count_returns_int(loaded_models): | |
| from llm_backend.tokenizer import token_count | |
| count = token_count("The cat sat on the mat") | |
| assert isinstance(count, int) | |
| assert count > 0 | |
| def test_token_count_empty_returns_zero(loaded_models): | |
| from llm_backend.tokenizer import token_count | |
| assert token_count("") == 0 | |
| assert token_count(" ") == 0 | |
| def test_token_count_less_than_word_count(loaded_models): | |
| from llm_backend.tokenizer import token_count | |
| text = "The cat sat on the mat" | |
| word_count = len(text.split()) | |
| count = token_count(text) | |
| assert count <= word_count or count >= word_count | |
| def test_get_context_window_split_short_text(loaded_models): | |
| from llm_backend.tokenizer import get_context_window_split | |
| result = get_context_window_split("The cat sat on the", context_window=1024) | |
| assert result["in_context"] == "The cat sat on the" | |
| assert result["out_of_context"] == "" | |
| assert result["total_tokens"] > 0 | |
| assert result["context_window"] == 1024 | |
| assert result["is_truncated"] is False | |
| def test_get_context_window_split_long_text(loaded_models): | |
| from llm_backend.tokenizer import get_context_window_split | |
| text = "word " * 500 | |
| result = get_context_window_split(text, context_window=100) | |
| assert result["is_truncated"] is True | |
| assert result["out_of_context"] != "" | |
| assert result["in_context"] != "" | |
| assert result["total_tokens"] > 100 | |
| def test_get_context_window_split_exact_boundary(loaded_models): | |
| from llm_backend.tokenizer import get_context_window_split, token_count | |
| text = "hello world" | |
| count = token_count(text) | |
| result = get_context_window_split(text, context_window=count) | |
| assert result["is_truncated"] is False | |
| assert result["in_context"] == text | |
| assert result["out_of_context"] == "" | |
| def test_get_context_window_split_empty_text(loaded_models): | |
| from llm_backend.tokenizer import get_context_window_split | |
| result = get_context_window_split("", context_window=1024) | |
| assert result["in_context"] == "" | |
| assert result["out_of_context"] == "" | |
| assert result["total_tokens"] == 0 | |
| assert result["is_truncated"] is False | |
| def test_get_context_window_split_returns_all_keys(loaded_models): | |
| from llm_backend.tokenizer import get_context_window_split | |
| result = get_context_window_split("test", context_window=10) | |
| assert "in_context" in result | |
| assert "out_of_context" in result | |
| assert "total_tokens" in result | |
| assert "context_window" in result | |
| assert "is_truncated" in result | |