TheAiCollectiveART's picture
Publish Zymatica Voice LLM hepta-architecture showcase codebases (part 2)
79e28b7 verified
Raw
History Blame Contribute Delete
4.52 kB
import argparse
# ==============================================================================
# COPY OF THE ACTUAL COMPRESSOR FUNCTIONS (compress_tokenizer.py)
# ==============================================================================
def write_varint(val):
res = bytearray()
while val >= 128:
res.append((val & 0x7F) | 0x80)
val >>= 7
res.append(val & 0x7F)
return bytes(res)
def get_prefix_suffix_encoding(tokens):
"""Encodes a list of token bytes using prefix-suffix compression."""
encoded = bytearray()
prev = b''
for t in tokens:
common = 0
l = min(len(t), len(prev))
while common < l and t[common] == prev[common]:
common += 1
suffix = t[common:]
encoded.extend(write_varint(common))
encoded.extend(write_varint(len(suffix)))
encoded.extend(suffix)
prev = t
return bytes(encoded)
# ==============================================================================
# DECODER IMPLEMENTATION FOR VERIFICATION
# ==============================================================================
def read_varint(data, pos):
val = 0
shift = 0
while True:
b = data[pos]
pos += 1
val |= (b & 0x7F) << shift
if not (b & 0x80):
break
shift += 7
return val, pos
def decode_prefix_suffix(encoded_bytes, num_tokens):
"""Losslessly decodes the prefix-suffix byte stream back to list of tokens."""
tokens = []
prev = b''
pos = 0
for _ in range(num_tokens):
common, pos = read_varint(encoded_bytes, pos)
suffix_len, pos = read_varint(encoded_bytes, pos)
suffix = encoded_bytes[pos : pos + suffix_len]
pos += suffix_len
# Reconstruct token: take common prefix from prev and append suffix
t = prev[:common] + suffix
tokens.append(t)
prev = t
return tokens
# ==============================================================================
def run_proof():
print("======================================================================")
print("ZYMATICA | Tokenizer Prefix-Suffix Varint Differential Coding Proof")
print("======================================================================\n")
# Sample vocabulary representing a lexicographically sorted tokenizer table
mock_vocab = [
"auth",
"author",
"authorities",
"authority",
"authorize",
"authorized",
"authorizing",
"auto",
"automate",
"automated",
"automatic",
"automation"
]
vocab_bytes = [t.encode('utf-8') for t in mock_vocab]
print("[1] Original Sorted Vocabulary:")
total_raw_bytes = 0
for idx, t in enumerate(mock_vocab):
raw_len = len(t)
total_raw_bytes += raw_len + 1 # 1 extra byte for string boundary/null terminator
print(f" ID {idx:2d}: '{t}'")
print(f" -> Total Uncompressed size (with boundaries): {total_raw_bytes} bytes")
print("\n[2] Executing Prefix-Suffix Varint Encoder...")
compressed_bytes = get_prefix_suffix_encoding(vocab_bytes)
compressed_len = len(compressed_bytes)
print(f" -> Encoded Binary Stream size: {compressed_len} bytes")
print(f" -> Binary Stream (Hex): {compressed_bytes.hex().upper()}")
print("\n[3] Executing Sequential Decoder Reassembly...")
decoded_bytes = decode_prefix_suffix(compressed_bytes, len(mock_vocab))
decoded_strings = [t.decode('utf-8') for t in decoded_bytes]
# Lossless validation checks
assert mock_vocab == decoded_strings, "Validation failed! Decoded strings do not match original."
print(" -> Lossless verification passed. Decoded strings are identical.")
compression_ratio = total_raw_bytes / compressed_len
savings = (1 - (compressed_len / total_raw_bytes)) * 100
print("\n[4] Summary Metrics:")
print(f" - Uncompressed size: {total_raw_bytes} bytes")
print(f" - Compressed size: {compressed_len} bytes")
print(f" - Space Savings: {savings:.2f}%")
print(f" - Compression Ratio: {compression_ratio:.2f}x")
print("\n[VERIFICATION] Tokenizer differential coder verified from actual codebase.")
if __name__ == "__main__":
parser = argparse.ArgumentParser(description="Zymatica Tokenizer Differential Coding Proof")
parser.add_argument("--test", action="store_true", help="Run test mode")
args = parser.parse_args()
run_proof()