Update README.md
#1
by AntonioTN - opened
README.md
CHANGED
|
@@ -411,6 +411,15 @@ If you use this model, please cite the base model and Pulsar 16B:
|
|
| 411 |
url = {https://huggingface.co/MultiverseComputingCAI/Pulsar-16B-BF16},
|
| 412 |
note = {Model developed based on nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16 using CompactifAI technology}
|
| 413 |
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 414 |
```
|
| 415 |
|
| 416 |
**Built by [Multiverse Computing](https://www.multiversecomputing.com)** · [Report an issue](TODO_PULSAR_HF_URL/discussions) · [Discord](https://discord.gg/cGas9uStqp)
|
|
|
|
| 411 |
url = {https://huggingface.co/MultiverseComputingCAI/Pulsar-16B-BF16},
|
| 412 |
note = {Model developed based on nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16 using CompactifAI technology}
|
| 413 |
}
|
| 414 |
+
@misc{ryskulov2026efficientknowledgedistillationllms,
|
| 415 |
+
title={Efficient Knowledge Distillation for LLMs: Offline Top-K Logits and a Fused Chunked KL Loss},
|
| 416 |
+
author={Bakbergen Ryskulov and Iker García-Ferrero and David Montero and David Jansen and Ali Hashemi and Jezabel R. Garcia and Antonio Tiene and Román Orús},
|
| 417 |
+
year={2026},
|
| 418 |
+
eprint={2608.03796},
|
| 419 |
+
archivePrefix={arXiv},
|
| 420 |
+
primaryClass={cs.CL},
|
| 421 |
+
url={https://arxiv.org/abs/2608.03796},
|
| 422 |
+
}
|
| 423 |
```
|
| 424 |
|
| 425 |
**Built by [Multiverse Computing](https://www.multiversecomputing.com)** · [Report an issue](TODO_PULSAR_HF_URL/discussions) · [Discord](https://discord.gg/cGas9uStqp)
|