Download reports/quantization.json from devildasdf/NEXORA: direct link, hf CLI and curl.
- Browser
- Download file 771 Bytes
-
https://huggingface.co/devildasdf/NEXORA/resolve/main/reports/quantization.json
- Command line
-
hf download hf://devildasdf/NEXORA/reports/quantization.json
-
curl -L -o quantization.json https://huggingface.co/devildasdf/NEXORA/resolve/main/reports/quantization.json
771 Bytes
| { | |
| "fp32": { | |
| "validation_loss": 3.2081077098846436, | |
| "batch_latency_seconds": { | |
| "p50": 0.029111100011505187, | |
| "p95": 0.03290360001847148, | |
| "p99": 0.03472879994660616 | |
| }, | |
| "samples": 20 | |
| }, | |
| "dynamic_int8_linear_only": { | |
| "validation_loss": 3.2121124267578125, | |
| "batch_latency_seconds": { | |
| "p50": 0.0270732999779284, | |
| "p95": 0.03375379997305572, | |
| "p99": 0.033887999947182834 | |
| }, | |
| "samples": 20 | |
| }, | |
| "max_logit_difference": 0.3234729766845703, | |
| "argmax_agreement": 0.96484375, | |
| "limitations": "Tiny model, one validation batch, CPU linear-layer INT8 only. Not INT8 embeddings, BF16/FP8/INT4 deployment, coding/reasoning/context/tool quality validation. Do not infer production speedup." | |
| } |