File size: 2,666 Bytes
46b9eea
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
{
  "release": "jiamingshan/AHA-L2A-Qwen3-1.7B-repro",
  "source_git": {
    "repository": "https://github.com/shanjiaming/AHA",
    "branch": "codex/l2a-router-granularity",
    "commit": "b47a549"
  },
  "base_checkpoint": {
    "location": "repository root",
    "description": "Qwen3-1.7B tuned vanilla, global step 6911",
    "sha256_model_safetensors": "22c971a65f7f1835b0e2a38b45f2f92191f5d8b63b30969f2f61a69d30afbc05"
  },
  "execution": {
    "distributed_training": false,
    "world_size_per_arm": 1,
    "gpu_count_per_arm": 1,
    "one_gpu_machine": "run AHA then L2A-style sequentially on GPU 0",
    "two_or_more_gpu_machine": "run AHA on GPU 0 and L2A-style on GPU 1 concurrently; remaining GPUs are unused",
    "arm_definition": "one independent experimental variant, not a processor architecture or GPU group"
  },
  "arms": {
    "aha": {
      "router_granularity": "token_kv_head",
      "description": "one native gate per token, KV head, and layer"
    },
    "l2a_style": {
      "router_granularity": "token",
      "description": "one native gate per token and layer, shared across all heads",
      "claim_boundary": "L2A-style shared-gate; not an official L2A reproduction"
    }
  },
  "common": {
    "local_attention": {
      "kind": "sink_recent",
      "sink_tokens": 64,
      "recent_tokens": 256
    },
    "gate_initialization": {
      "weight": "zeros",
      "full_probability": 0.9
    },
    "effective_sparsity_denominator": "token x KV-head x layer"
  },
  "stage_1": {
    "dataset": "recipe/data/am_distilled_long_mix",
    "rows": 1024,
    "max_length": 8192,
    "batch_size": 1,
    "optimizer_steps": 300,
    "epochs": 0.29296875,
    "trainable": "native gate rows only",
    "learning_rate": 0.00003,
    "warmup_ratio": 0.1,
    "hidden_state_distillation": 1.0,
    "regularizer": 0.1,
    "ce": 0.0,
    "train_threshold": 0.5,
    "seed": 42
  },
  "stage_2": {
    "dataset": "same 1024 rows",
    "max_length": 8192,
    "batch_size": 1,
    "optimizer_steps_run": 75,
    "selected_checkpoint_step": 25,
    "selected_epochs": 0.0244140625,
    "gate_learning_rate": 0.000003,
    "backbone_learning_rate": 0.0000003,
    "ce": 1.0,
    "attention_distillation": 0.5,
    "regularizer": 0.01,
    "train_threshold": 0.58,
    "freeze_embeddings_and_lm_head": true,
    "seed": 47
  },
  "evaluation": {
    "thresholds": [0.45, 0.5, 0.525, 0.55, 0.575, 0.6, 0.625, 0.65],
    "strict_prefill_and_decode_sparse_routing": true,
    "force_full_heads": false,
    "full_decode": false,
    "headline": "highest measured effective sparsity whose score is at least 95% of tuned vanilla on every suite"
  }
}