{ "status": "VALIDATED_TOY_OPTIMIZATION_ONLY", "base": "artifacts/tiny", "rank": 4, "alpha": 8, "targets": [ "blocks.0.attn.q", "blocks.0.attn.v", "blocks.1.attn.q", "blocks.1.attn.v", "blocks.2.attn.q", "blocks.2.attn.v", "blocks.3.attn.q", "blocks.3.attn.v" ], "seconds": 2.3280960000120103, "history": [ { "stage": "SFT", "step": 1, "loss": 2.9821512699127197 }, { "stage": "SFT", "step": 11, "loss": 3.0910983085632324 }, { "stage": "SFT", "step": 21, "loss": 2.4964144229888916 }, { "stage": "SFT", "step": 31, "loss": 2.724806308746338 }, { "stage": "SFT", "step": 40, "loss": 2.5710480213165283 }, { "stage": "DPO", "step": 1, "loss": 0.09377383440732956 }, { "stage": "DPO", "step": 2, "loss": 0.06439569592475891 }, { "stage": "DPO", "step": 3, "loss": 0.04369953274726868 }, { "stage": "DPO", "step": 4, "loss": 0.03653418645262718 }, { "stage": "DPO", "step": 5, "loss": 0.031024407595396042 }, { "stage": "DPO", "step": 6, "loss": 0.024815931916236877 }, { "stage": "DPO", "step": 7, "loss": 0.018242638558149338 }, { "stage": "DPO", "step": 8, "loss": 0.01337386667728424 }, { "stage": "DPO", "step": 9, "loss": 0.010624001733958721 }, { "stage": "DPO", "step": 10, "loss": 0.008802279829978943 } ], "merge_max_abs_error": 2.5033950805664062e-06, "limitations": "Four synthetic SFT examples and one preference pair; no evidence of reasoning improvement or generalization. Adapter not enabled by default." }