File size: 503 Bytes
f7df3ab
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
{
  "repo_id": "Ethosoft/Qwen3-1.7B-ResearchReasoning-JSON-RL",
  "base_model": "Qwen/Qwen3-1.7B",
  "artifact_type": "PEFT LoRA adapter",
  "project_stage_name": "RL-lite",
  "method_note": "Verifier-guided rejection sampling followed by supervised fine-tuning; not full online policy-gradient RL.",
  "fine_tuning_max_sequence_length": 4096,
  "languages": ["en", "tr"],
  "github": "https://github.com/Ahmet2001/QA-research-SLM",
  "gguf_repo": "Ethosoft/Qwen3-1.7B-ResearchReasoning-JSON-RL-GGUF"
}