import os import json import torch import numpy as np import gguf from safetensors.torch import load_file import sentencepiece as spm from pathlib import Path def permute_rope(w, n_heads=8, head_dim=48): # Permute weights from HF half-half RoPE layout to llama.cpp interleaved RoPE layout return w.view(n_heads, 2, head_dim // 2, -1).transpose(1, 2).reshape(w.shape) def export_gguf(weights_path="exported_model/model.safetensors", tokenizer_path="tokenizer/tokenizer.model", quant_type=gguf.GGMLQuantizationType.F16, output_file="webgpu_space/model_f16.gguf"): print(f"Exporting permuted GGUF model: {output_file} with quant type: {quant_type.name}...") weights = load_file(weights_path) if "tok_embeddings.weight" not in weights and "output.weight" in weights: weights["tok_embeddings.weight"] = weights["output.weight"] sp = spm.SentencePieceProcessor(model_file=tokenizer_path) vocab_size = sp.get_piece_size() tokens = [] scores = [] tok_types = [] for i in range(vocab_size): tokens.append(sp.id_to_piece(i)) scores.append(sp.get_score(i)) if sp.is_unknown(i): tok_types.append(gguf.TokenType.UNKNOWN) elif sp.is_control(i): tok_types.append(gguf.TokenType.CONTROL) elif sp.is_byte(i): tok_types.append(gguf.TokenType.BYTE) else: tok_types.append(gguf.TokenType.NORMAL) writer = gguf.GGUFWriter(output_file, "llama") writer.add_name("Simple-Stories-Hindi-20M") writer.add_context_length(512) writer.add_embedding_length(384) writer.add_feed_forward_length(1024) writer.add_block_count(10) writer.add_head_count(8) writer.add_head_count_kv(8) writer.add_rope_dimension_count(48) writer.add_rope_freq_base(10000.0) writer.add_layer_norm_rms_eps(1e-5) writer.add_vocab_size(vocab_size) writer.add_tokenizer_model("llama") writer.add_token_list(tokens) writer.add_token_scores(scores) writer.add_token_types(tok_types) writer.add_bos_token_id(2) writer.add_eos_token_id(3) writer.add_pad_token_id(0) writer.add_unk_token_id(1) mapping = { "tok_embeddings.weight": "token_embd.weight", "norm.weight": "output_norm.weight", "output.weight": "output.weight", } for i in range(10): mapping[f"layers.{i}.attention.wq.weight"] = f"blk.{i}.attn_q.weight" mapping[f"layers.{i}.attention.wk.weight"] = f"blk.{i}.attn_k.weight" mapping[f"layers.{i}.attention.wv.weight"] = f"blk.{i}.attn_v.weight" mapping[f"layers.{i}.attention.wo.weight"] = f"blk.{i}.attn_output.weight" mapping[f"layers.{i}.attention_norm.weight"] = f"blk.{i}.attn_norm.weight" mapping[f"layers.{i}.feed_forward.w1.weight"] = f"blk.{i}.ffn_gate.weight" mapping[f"layers.{i}.feed_forward.w2.weight"] = f"blk.{i}.ffn_up.weight" mapping[f"layers.{i}.feed_forward.w3.weight"] = f"blk.{i}.ffn_down.weight" mapping[f"layers.{i}.ffn_norm.weight"] = f"blk.{i}.ffn_norm.weight" for orig_name, gguf_name in mapping.items(): tensor = weights[orig_name] # Permute WQ and WK weights for llama.cpp RoPE layout! if "attn_q.weight" in gguf_name or "attn_k.weight" in gguf_name: tensor = permute_rope(tensor, n_heads=8, head_dim=48) tensor_np = tensor.numpy().astype(np.float32) if tensor_np.ndim == 2 and quant_type != gguf.GGMLQuantizationType.F16: quant_data = gguf.quantize(tensor_np, quant_type) writer.add_tensor(gguf_name, quant_data, raw_dtype=quant_type) else: if quant_type == gguf.GGMLQuantizationType.F16 and tensor_np.ndim == 2: writer.add_tensor(gguf_name, tensor_np.astype(np.float16)) else: writer.add_tensor(gguf_name, tensor_np) writer.write_header_to_file() writer.write_kv_data_to_file() writer.write_tensors_to_file() writer.close() file_size_mb = os.path.getsize(output_file) / (1024 * 1024) print(f"Successfully generated permuted {output_file} ({file_size_mb:.2f} MB)!") def generate_tokenizer_json(tokenizer_path="tokenizer/tokenizer.model", output_file="webgpu_space/tokenizer.json"): sp = spm.SentencePieceProcessor(model_file=tokenizer_path) tokenizer_json = { "version": "1.0", "truncation": None, "padding": None, "added_tokens": [ {"id": 0, "special": True, "content": "", "single_word": False, "lstrip": False, "rstrip": False, "normalized": False}, {"id": 1, "special": True, "content": "", "single_word": False, "lstrip": False, "rstrip": False, "normalized": False}, {"id": 2, "special": True, "content": "", "single_word": False, "lstrip": False, "rstrip": False, "normalized": False}, {"id": 3, "special": True, "content": "", "single_word": False, "lstrip": False, "rstrip": False, "normalized": False} ], "normalizer": None, "pre_tokenizer": None, "post_processor": None, "decoder": None, "model": { "type": "Unigram", "vocab": [[sp.id_to_piece(i), sp.get_score(i)] for i in range(sp.get_piece_size())] } } with open(output_file, "w", encoding="utf-8") as f: json.dump(tokenizer_json, f, ensure_ascii=False, indent=2) print(f"Successfully generated {output_file}!") if __name__ == "__main__": export_gguf(quant_type=gguf.GGMLQuantizationType.F16, output_file="webgpu_space/model_f16.gguf") export_gguf(quant_type=gguf.GGMLQuantizationType.Q8_0, output_file="webgpu_space/model_q8_0.gguf") generate_tokenizer_json()