File size: 4,171 Bytes
35ea7a1
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
# ==============================================================================
# 🚀 ViuTranslate — Evaluation & Benchmark Suite
# ==============================================================================

import os
import sys
import json
import time
import torch
from tokenizers import Tokenizer
from huggingface_hub import hf_hub_download

if hasattr(sys.stdout, "reconfigure"):
    sys.stdout.reconfigure(encoding="utf-8", errors="replace")

cur_dir = os.path.dirname(os.path.abspath(__file__)) if "__file__" in locals() else os.getcwd()
if cur_dir not in sys.path:
    sys.path.insert(0, cur_dir)

from model import ViuAI
from config import ViuAIConfig

REPO_ID = "ViuAI/ViuTranslate"
DATA_REPO_ID = "ViuAI/ViuTranslate-Data"
EOT_ID = 64002

def main():
    print("=" * 80)
    print("📊 ViuTranslate Benchmark Evaluation (Gold Human Test Set)")
    print("=" * 80)
    
    device = torch.device("cuda" if torch.cuda.is_available() else "cpu")
    print(f"Device: {device.type.upper()}")
    
    # Load Model
    ckpt_path = "viutranslate_final.pt"
    if not os.path.exists(ckpt_path):
        ckpt_path = hf_hub_download(repo_id=REPO_ID, filename="viutranslate_final.pt")
        
    tok_path = "tokenizer.json"
    if not os.path.exists(tok_path):
        tok_path = hf_hub_download(repo_id=REPO_ID, filename="tokenizer.json")
        
    tokenizer = Tokenizer.from_file(tok_path)
    cfg = ViuAIConfig(vocab_size=64003, context_length=2048)
    model = ViuAI(cfg).to(device)
    
    state = torch.load(ckpt_path, map_location=device, weights_only=False)
    weights = state.get("model_state_dict", state)
    model.load_state_dict(weights, strict=False)
    model.eval()
    print("✅ Model loaded.")
    
    # Download Validation Set
    val_json_path = "raw/viu_translate_val.json"
    if not os.path.exists(val_json_path):
        try:
            val_json_path = hf_hub_download(repo_id=DATA_REPO_ID, filename="raw/viu_translate_val.json", repo_type="dataset")
        except Exception:
            val_json_path = None
            
    if val_json_path and os.path.exists(val_json_path):
        with open(val_json_path, 'r', encoding='utf-8') as f:
            val_data = json.load(f)
        print(f"Loaded {len(val_data):,} gold validation pairs.")
    else:
        print("Val JSON not found. Running curated benchmark cases.")
        val_data = []

    # Curated Benchmark Tests
    benchmark_cases = [
        ("The sun rises in the east and sets in the west.", "सूरज पूर्व में उगता है और पश्चिम में डूबता है।"),
        ("Consistency and discipline are the keys to long term success.", "निरंतरता और अनुशासन दीर्घकालिक सफलता की कुंजी हैं।"),
        ("Artificial intelligence is transforming industries across the globe.", "कृत्रिम बुद्धिमत्ता दुनिया भर के उद्योगों को बदल रही है।"),
        ("Where is the nearest railway station?", "निकटतम रेलवे स्टेशन कहाँ है?"),
        ("Regular exercise is essential for maintaining physical and mental health.", "शारीरिक और मानसिक स्वास्थ्य बनाए रखने के लिए नियमित व्यायाम आवश्यक है।")
    ]
    
    print("\n🔍 Running Curated Test Cases:")
    for en, hi_ref in benchmark_cases:
        prompt = f"<|user|>\n{en}<|endofturn|>\n<|assistant|>\n"
        inp = torch.tensor([tokenizer.encode(prompt).ids], device=device)
        with torch.no_grad():
            out = model.generate(inp, max_new_tokens=80, temperature=0.2, eos_token_id=EOT_ID)
        gen = tokenizer.decode(out[0][inp.shape[1]:].tolist()).replace("<|endofturn|>", "").strip()
        print(f"\n• EN:  {en}")
        print(f"  REF: {hi_ref}")
        print(f"  GEN: {gen}")

    print("\n" + "=" * 80)
    print("✅ Evaluation Completed!")
    print("=" * 80)

if __name__ == "__main__":
    main()