File size: 6,754 Bytes
b3c2a26
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
5a98e33
 
 
 
b3c2a26
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
5a98e33
 
 
 
 
 
b3c2a26
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
5a98e33
 
b3c2a26
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
"""Release validation for KorByte-128K artifacts."""

from __future__ import annotations

import hashlib
import json
import random
from datetime import UTC, datetime
from pathlib import Path
from typing import Any

from tokenizers import Tokenizer
from transformers import PreTrainedTokenizerFast

from .challenges import CHALLENGE_TEXTS
from .config import CORE_VOCAB_SIZE, TOTAL_VOCAB_SIZE, Paths
from .special_tokens import special_token_strings

REQUIRED_RELEASE_FILES = (
    "README.md",
    "LICENSE",
    "NOTICE",
    "DATA_SOURCES.md",
    "tokenizer.json",
    "tokenizer_config.json",
    "special_tokens_map.json",
    "added_tokens.json",
    "vocab.json",
    "merges.txt",
    "provenance/corpus_manifest.json",
    "provenance/build_manifest.json",
    "reports/benchmark.json",
    "reports/benchmark.md",
    "reports/comparison.json",
    "reports/comparison.md",
    "reports/research.json",
    "reports/research.md",
)


def _sha256(path: Path) -> str:
    digest = hashlib.sha256()
    with path.open("rb") as handle:
        for chunk in iter(lambda: handle.read(1024 * 1024), b""):
            digest.update(chunk)
    return digest.hexdigest()


def _random_challenges(*, count: int = 500, seed: int = 20260804) -> list[str]:
    rng = random.Random(seed)
    alphabets = (
        tuple(chr(codepoint) for codepoint in range(0xAC00, 0xD7A4, 173)),
        tuple(chr(codepoint) for codepoint in range(0x1100, 0x1200, 7)),
        tuple(chr(codepoint) for codepoint in range(0x4E00, 0x9FFF, 997)),
        tuple(chr(codepoint) for codepoint in range(0x1F300, 0x1FAFF, 47)),
        tuple(chr(codepoint) for codepoint in range(0x0300, 0x0370, 5)),
        tuple("abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789"),
        tuple(" .,!?;:'\"()[]{}<>+-=*/_@#%&|~`\t\n\r"),
    )
    strings: list[str] = []
    for _ in range(count):
        length = rng.randint(0, 160)
        value = "".join(rng.choice(rng.choice(alphabets)) for _ in range(length))
        strings.append(value)
    return strings


def _assert_round_trip(tokenizer: Tokenizer, texts: list[str]) -> int:
    tested = 0
    for text in texts:
        encoding = tokenizer.encode(text, add_special_tokens=False)
        decoded = tokenizer.decode(encoding.ids, skip_special_tokens=False)
        if decoded != text:
            raise AssertionError(
                f"Tokenizers round-trip failed at case {tested}: {text!r} != {decoded!r}"
            )
        if any(token_id < 0 or token_id >= TOTAL_VOCAB_SIZE for token_id in encoding.ids):
            raise AssertionError(f"Out-of-range token ID at case {tested}")
        tested += 1
    return tested


def validate_release(root: Path, *, require_benchmark_gate: bool = True) -> dict[str, Any]:
    """Validate packaging, ID allocation, round-trip behavior, and benchmark gate."""

    missing = [name for name in REQUIRED_RELEASE_FILES if not (root / name).is_file()]
    if missing:
        raise FileNotFoundError(f"Missing release files: {', '.join(missing)}")

    tokenizer_path = root / "tokenizer.json"
    tokenizer = Tokenizer.from_file(str(tokenizer_path))
    core_size = tokenizer.get_vocab_size(with_added_tokens=False)
    total_size = tokenizer.get_vocab_size(with_added_tokens=True)
    if core_size != CORE_VOCAB_SIZE:
        raise AssertionError(f"Core vocabulary is {core_size}, expected {CORE_VOCAB_SIZE}")
    if total_size != TOTAL_VOCAB_SIZE:
        raise AssertionError(f"Total vocabulary is {total_size}, expected {TOTAL_VOCAB_SIZE}")
    if tokenizer.normalizer is not None:
        raise AssertionError("A normalizer would prevent exact code-point preservation")

    vocab = tokenizer.get_vocab(with_added_tokens=True)
    special_ids = [vocab[token] for token in special_token_strings()]
    expected_special_ids = list(range(CORE_VOCAB_SIZE, TOTAL_VOCAB_SIZE))
    if special_ids != expected_special_ids:
        raise AssertionError("Special-token IDs are not stable and contiguous")

    fixed_texts = list(CHALLENGE_TEXTS)
    random_texts = _random_challenges()
    tokenizers_cases = _assert_round_trip(tokenizer, fixed_texts + random_texts)

    fast = PreTrainedTokenizerFast.from_pretrained(str(root), local_files_only=True)
    transformers_cases = 0
    for text in fixed_texts + random_texts[:100]:
        token_ids = fast.encode(text, add_special_tokens=False)
        decoded = fast.decode(
            token_ids,
            skip_special_tokens=False,
            clean_up_tokenization_spaces=False,
        )
        if decoded != text:
            raise AssertionError(
                f"Transformers round-trip failed at case {transformers_cases}: "
                f"{text!r} != {decoded!r}"
            )
        transformers_cases += 1

    paths = Paths(root)
    benchmark = json.loads(paths.benchmark_json.read_text(encoding="utf-8"))
    benchmark_passed = bool(benchmark.get("compression_gate_passed"))
    if require_benchmark_gate and not benchmark_passed:
        reduction = benchmark.get("korbyte_macro_reduction_vs_kanana_percent")
        raise AssertionError(f"Compression gate failed: macro reduction={reduction!r}%")

    comparison = json.loads(paths.comparison_json.read_text(encoding="utf-8"))
    comparison_passed = bool(comparison.get("first_place_gate_passed"))
    comparison_coverage_passed = bool(comparison.get("comparison_coverage_gate_passed"))
    if require_benchmark_gate and not (comparison_passed and comparison_coverage_passed):
        raise AssertionError("Pinned public first-place comparison gate failed")

    build_manifest = json.loads(paths.build_manifest.read_text(encoding="utf-8"))
    current_hash = _sha256(tokenizer_path)
    if build_manifest.get("tokenizer_sha256") != current_hash:
        raise AssertionError("tokenizer.json does not match the build manifest")

    result = {
        "schema_version": 1,
        "created_at": datetime.now(UTC).isoformat(),
        "passed": True,
        "core_vocab_size": core_size,
        "total_vocab_size": total_size,
        "special_token_ids_contiguous": True,
        "tokenizers_round_trip_cases": tokenizers_cases,
        "transformers_round_trip_cases": transformers_cases,
        "benchmark_gate_required": require_benchmark_gate,
        "benchmark_gate_passed": benchmark_passed,
        "comparison_gate_passed": comparison_passed,
        "comparison_coverage_gate_passed": comparison_coverage_passed,
        "tokenizer_sha256": current_hash,
        "required_release_files": list(REQUIRED_RELEASE_FILES),
    }
    paths.validation_json.parent.mkdir(parents=True, exist_ok=True)
    paths.validation_json.write_text(
        json.dumps(result, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
    )
    return result