Publish metadata-complete Gemma-4 E2B-it text decoder (fp16 CUDA): canonical inference_metadata.yaml + policies + L4/L5 parity; supersedes prior metadata-less multimodal export
23835eb verified | { | |
| "onnxruntime": "1.27.0 (onnxruntime-gpu)", | |
| "execution_provider": "CUDAExecutionProvider (NVIDIA H200)", | |
| "onnx_ir": "1.0.0", | |
| "onnxscript": "0.7.1", | |
| "transformers": "5.14.1", | |
| "torch": "2.13.0", | |
| "python": "3.12", | |
| "dtype": "float16", | |
| "mobius_git_sha": "0776f562daeefad468277177b1e92abb26486355", | |
| "build_flags": "--features text-only --dtype f16 --ep cuda --optimize=group_query_attention,packed_attention,skip_norm", | |
| "package_kind": "target_decoder", | |
| "canonical_metadata": "inference_metadata.yaml (schema 1.0, hashless)" | |
| } |