{ "tag": "gpt2g", "repo": "onnx-community/gpt2-ONNX-GQA", "kind": "text-gqa", "notas": "GPT-2 com atencao agrupada (GQA) exportada ONNX/OGA", "matrix": "transformer.wte.weight", "shape": [ 50257, 768 ], "dtype": "1", "K_experts": 32, "K_xi": 8, "d_src": 768, "n_adapters": 4, "inertia": 4229.31640625, "samples": 6000, "int4": true, "adapter_rank": 48, "epoca": "<= 2024/2025", "gqa_signature": { "gqa_nodes": 12, "ops_top": [ [ "Add", 97 ], [ "MatMul", 73 ], [ "LayerNormalization", 25 ], [ "GroupQueryAttention", 12 ], [ "Gelu", 12 ], [ "Gather", 3 ], [ "Constant", 2 ], [ "Cast", 2 ] ], "kv_candidate_shapes": [] } }