{ "repo_id": "Ethosoft/Qwen3-1.7B-ResearchReasoning-JSON-RL", "base_model": "Qwen/Qwen3-1.7B", "artifact_type": "PEFT LoRA adapter", "project_stage_name": "RL-lite", "method_note": "Verifier-guided rejection sampling followed by supervised fine-tuning; not full online policy-gradient RL.", "fine_tuning_max_sequence_length": 4096, "languages": ["en", "tr"], "github": "https://github.com/Ahmet2001/QA-research-SLM", "gguf_repo": "Ethosoft/Qwen3-1.7B-ResearchReasoning-JSON-RL-GGUF" }