NTQuoc
/

OpenRS-GRPO

@@ -1,9 +1,11 @@
 ---
 base_model: Qwen/Qwen3.5-0.8B
 library_name: transformers
 model_name: OpenRS-GRPO
 tags:
 - generated_from_trainer
 - trl
 - grpo
 licence: license
@@ -11,7 +13,7 @@ licence: license
 # Model Card for OpenRS-GRPO
-This model is a fine-tuned version of [Qwen/Qwen3.5-0.8B](https://huggingface.co/Qwen/Qwen3.5-0.8B).
 It has been trained using [TRL](https://github.com/huggingface/trl).
 ## Quick start

 ---
 base_model: Qwen/Qwen3.5-0.8B
+datasets: knoveleng/open-rs
 library_name: transformers
 model_name: OpenRS-GRPO
 tags:
 - generated_from_trainer
+- open-r1
 - trl
 - grpo
 licence: license
 # Model Card for OpenRS-GRPO
+This model is a fine-tuned version of [Qwen/Qwen3.5-0.8B](https://huggingface.co/Qwen/Qwen3.5-0.8B) on the [knoveleng/open-rs](https://huggingface.co/datasets/knoveleng/open-rs) dataset.
 It has been trained using [TRL](https://github.com/huggingface/trl).
 ## Quick start

config.json CHANGED Viewed

@@ -1,62 +1,72 @@
 {
-  "architectures": [
-    "Qwen2ForCausalLM"
-  ],
   "attention_dropout": 0.0,
-  "bos_token_id": 151646,
-  "dtype": "float16",
-  "eos_token_id": 151643,
   "hidden_act": "silu",
-  "hidden_size": 1536,
   "initializer_range": 0.02,
-  "intermediate_size": 8960,
   "layer_types": [
     "full_attention",
     "full_attention",
     "full_attention",
     "full_attention",
     "full_attention",
-    "full_attention",
-    "full_attention",
-    "full_attention",
-    "full_attention",
-    "full_attention",
-    "full_attention",
-    "full_attention",
-    "full_attention",
-    "full_attention",
-    "full_attention",
-    "full_attention",
-    "full_attention",
-    "full_attention",
-    "full_attention",
-    "full_attention",
-    "full_attention",
-    "full_attention",
-    "full_attention",
-    "full_attention",
-    "full_attention",
-    "full_attention",
-    "full_attention",
     "full_attention"
   ],
-  "max_position_embeddings": 131072,
-  "max_window_layers": 21,
-  "model_type": "qwen2",
-  "num_attention_heads": 12,
-  "num_hidden_layers": 28,
   "num_key_value_heads": 2,
-  "pad_token_id": 151643,
   "rms_norm_eps": 1e-06,
   "rope_parameters": {
-    "rope_theta": 10000,
     "rope_type": "default"
   },
-  "sliding_window": null,
-  "tie_word_embeddings": false,
   "transformers_version": "5.8.0.dev0",
   "use_cache": true,
-  "use_mrope": false,
-  "use_sliding_window": false,
-  "vocab_size": 151936
 }

 {
+  "attention_bias": false,
   "attention_dropout": 0.0,
+  "attn_output_gate": true,
+  "bos_token_id": null,
+  "dtype": "bfloat16",
+  "eos_token_id": 248046,
+  "full_attention_interval": 4,
+  "head_dim": 256,
   "hidden_act": "silu",
+  "hidden_size": 1024,
   "initializer_range": 0.02,
+  "intermediate_size": 3584,
   "layer_types": [
+    "linear_attention",
+    "linear_attention",
+    "linear_attention",
     "full_attention",
+    "linear_attention",
+    "linear_attention",
+    "linear_attention",
     "full_attention",
+    "linear_attention",
+    "linear_attention",
+    "linear_attention",
     "full_attention",
+    "linear_attention",
+    "linear_attention",
+    "linear_attention",
     "full_attention",
+    "linear_attention",
+    "linear_attention",
+    "linear_attention",
     "full_attention",
+    "linear_attention",
+    "linear_attention",
+    "linear_attention",
     "full_attention"
   ],
+  "linear_conv_kernel_dim": 4,
+  "linear_key_head_dim": 128,
+  "linear_num_key_heads": 16,
+  "linear_num_value_heads": 16,
+  "linear_value_head_dim": 128,
+  "mamba_ssm_dtype": "float32",
+  "max_position_embeddings": 262144,
+  "mlp_only_layers": [],
+  "model_type": "qwen3_5_text",
+  "mtp_num_hidden_layers": 1,
+  "mtp_use_dedicated_embeddings": false,
+  "num_attention_heads": 8,
+  "num_hidden_layers": 24,
   "num_key_value_heads": 2,
+  "pad_token_id": 248044,
+  "partial_rotary_factor": 0.25,
   "rms_norm_eps": 1e-06,
   "rope_parameters": {
+    "mrope_interleaved": true,
+    "mrope_section": [
+      11,
+      11,
+      10
+    ],
+    "partial_rotary_factor": 0.25,
+    "rope_theta": 10000000,
     "rope_type": "default"
   },
+  "tie_word_embeddings": true,
   "transformers_version": "5.8.0.dev0",
   "use_cache": true,
+  "vocab_size": 248320
 }