| """ |
| Frox AI Morph 1.1 — Tokenizer Tests |
| Run with: pytest tests/test_tokenizer.py -v |
| |
| Uses the synthetic-corpus fallback path (no network needed) so this |
| suite runs offline. Builds a small real BPE tokenizer once per test |
| session rather than mocking it — the whole point is to catch real |
| tokenizer-training regressions (e.g. a special token silently missing |
| an ID, or the chat template drifting out of sync with the parser in |
| morph_engine.py). |
| """ |
| from __future__ import annotations |
|
|
| import sys |
| from pathlib import Path |
|
|
| import pytest |
|
|
| sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) |
|
|
| from tokenizer.morph_tokenizer import ( |
| ALL_SPECIAL_TOKENS, MORPH_SPECIAL_TOKENS, |
| apply_chat_template, build_morph_tokenizer, |
| format_dpo_pair, format_thinking, format_tool_call, format_tool_result, |
| _make_synthetic_corpus, _train_bpe, _wrap_tokenizer, |
| ) |
|
|
|
|
| @pytest.fixture(scope="session") |
| def tokenizer(tmp_path_factory): |
| """Build one small tokenizer for the whole test session (BPE training isn't free).""" |
| corpus_dir = tmp_path_factory.mktemp("corpus") |
| corpus_path = _make_synthetic_corpus(corpus_dir) |
| tok_model = _train_bpe([corpus_path], vocab_size=800) |
| return _wrap_tokenizer(tok_model) |
|
|
|
|
| class TestSpecialTokens: |
| def test_all_special_tokens_present(self, tokenizer): |
| vocab = tokenizer.get_vocab() |
| missing = [t for t in ALL_SPECIAL_TOKENS if t not in vocab] |
| assert not missing, f"Missing special tokens: {missing}" |
|
|
| def test_morph_specific_tokens_have_unique_ids(self, tokenizer): |
| vocab = tokenizer.get_vocab() |
| ids = [vocab[t] for t in MORPH_SPECIAL_TOKENS] |
| assert len(ids) == len(set(ids)), "Two Morph special tokens collapsed to the same ID" |
|
|
| def test_core_attributes_set(self, tokenizer): |
| assert tokenizer.pad_token == "<|pad|>" |
| assert tokenizer.bos_token == "<|begin_of_text|>" |
| assert tokenizer.eos_token == "<|end_of_text|>" |
| assert tokenizer.pad_token_id is not None |
| assert tokenizer.eos_token_id is not None |
|
|
|
|
| class TestEncodeDecodeRoundTrip: |
| def test_simple_text_round_trips(self, tokenizer): |
| text = "Neural networks compute features through multiple layers." |
| ids = tokenizer.encode(text, add_special_tokens=False) |
| decoded = tokenizer.decode(ids, skip_special_tokens=True) |
| |
| |
| assert "Neural" in decoded or "neural" in decoded.lower() |
| assert len(ids) > 0 |
|
|
| def test_empty_string_does_not_crash(self, tokenizer): |
| ids = tokenizer.encode("", add_special_tokens=False) |
| assert isinstance(ids, list) |
|
|
|
|
| class TestChatTemplate: |
| def test_produces_expected_structure(self, tokenizer): |
| messages = [{"role": "user", "content": "Hello!"}] |
| prompt = apply_chat_template(messages, tokenizer, add_generation_prompt=True) |
|
|
| assert "<|begin_of_text|>" in prompt |
| assert "<|start_header_id|>user<|end_header_id|>" in prompt |
| assert "Hello!" in prompt |
| assert prompt.rstrip().endswith( |
| "<|start_header_id|>assistant<|end_header_id|>" |
| ), "Generation prompt (assistant header) must be the last thing in the prompt" |
|
|
| def test_system_prompt_injected_first(self, tokenizer): |
| messages = [{"role": "user", "content": "Hi"}] |
| prompt = apply_chat_template( |
| messages, tokenizer, add_generation_prompt=True, |
| system_prompt="You are a helpful assistant.", |
| ) |
| system_pos = prompt.find("system") |
| user_pos = prompt.find("user") |
| assert 0 <= system_pos < user_pos, "System turn must come before the user turn" |
|
|
| def test_existing_system_message_not_duplicated(self, tokenizer): |
| messages = [ |
| {"role": "system", "content": "Custom system prompt"}, |
| {"role": "user", "content": "Hi"}, |
| ] |
| prompt = apply_chat_template( |
| messages, tokenizer, add_generation_prompt=True, |
| system_prompt="This should NOT appear since messages[0] is already system", |
| ) |
| assert prompt.count("<|start_header_id|>system<|end_header_id|>") == 1 |
| assert "Custom system prompt" in prompt |
| assert "This should NOT appear" not in prompt |
|
|
| def test_multi_turn_preserves_order(self, tokenizer): |
| messages = [ |
| {"role": "user", "content": "First question"}, |
| {"role": "assistant", "content": "First answer"}, |
| {"role": "user", "content": "Second question"}, |
| ] |
| prompt = apply_chat_template(messages, tokenizer, add_generation_prompt=True) |
| assert prompt.find("First question") < prompt.find("First answer") |
| assert prompt.find("First answer") < prompt.find("Second question") |
|
|
|
|
| class TestToolFormatting: |
| def test_tool_call_format_is_parseable_json(self, tokenizer): |
| import re, json |
| formatted = format_tool_call("web_search", {"query": "test"}) |
| match = re.search(r"<\|tool_call\|>(.*?)<\|/tool_call\|>", formatted) |
| assert match is not None |
| parsed = json.loads(match.group(1)) |
| assert parsed["name"] == "web_search" |
| assert parsed["args"]["query"] == "test" |
|
|
| def test_tool_result_format_is_parseable_json(self, tokenizer): |
| import re, json |
| formatted = format_tool_result("web_search", {"results": [1, 2, 3]}) |
| match = re.search(r"<\|tool_result\|>(.*?)<\|/tool_result\|>", formatted) |
| assert match is not None |
| parsed = json.loads(match.group(1)) |
| assert parsed["name"] == "web_search" |
|
|
| def test_thinking_block_wraps_content(self): |
| formatted = format_thinking("Let me reason step by step.") |
| assert formatted == "<|think|>Let me reason step by step.<|/think|>" |
|
|
|
|
| class TestDPOFormatting: |
| def test_prompt_chosen_rejected_share_prefix(self, tokenizer): |
| messages = [{"role": "user", "content": "What's 2+2?"}] |
| pair = format_dpo_pair( |
| messages, chosen="4", rejected="I don't know", tokenizer=tokenizer, |
| ) |
| assert pair["chosen"].startswith(pair["prompt"]) |
| assert pair["rejected"].startswith(pair["prompt"]) |
| assert pair["chosen"] != pair["rejected"] |
| assert "4" in pair["chosen"] |
| assert "I don't know" in pair["rejected"] |
|
|