Project Uroboros: Stage 2 — ARC

Browse files

Files changed (9) hide show

README.md +49 -0
chat_template.jinja +1 -0
merges.txt +0 -0
roste_adapter.pt +3 -0
roste_config.json +25 -0
special_tokens_map.json +49 -0
tokenizer.json +0 -0
tokenizer_config.json +169 -0
vocab.json +0 -0

README.md ADDED Viewed

	@@ -0,0 +1,49 @@

+---
+license: apache-2.0
+base_model: ByteDance/Ouro-2.6B-Thinking
+tags:
+  - roste
+  - lora
+  - 4bit
+  - arc-reasoning
+  - stage2
+---
+# Project Uroboros — RoSTE Stage 2 Adapter
+**Stage 2 of 2** in the Uroboros continual-learning pipeline.
+## Training details
+| Setting | Value |
+|---|---|
+| Base model | `ByteDance/Ouro-2.6B-Thinking` |
+| Datasets | UltraChat-200k → NVARC-augmented-puzzles |
+| Quantization | 4-bit NF4 double-quant (BitsAndBytes) |
+| LoRA r / alpha | 16 / 32 |
+| Bits (w / a / kv) | 4 / 4 / 4 |
+| Rotations used | R3 (Q/K head_dim), R4 (down_proj input, block-Hadamard) |
+| Max seq length | 1024 |
+| Transformers | `4.54.1` |
+## Files
+| File | Description |
+|---|---|
+| `roste_adapter.pt` | Trained `lora_A` + `lora_B` tensors from `apply_roste` |
+| `roste_config.json` | Hyperparameters needed to reconstruct the model |
+| `tokenizer.*` | Tokenizer saved from base model |
+## How to load
+See the [project notebook](https://www.kaggle.com/) for the full `apply_roste` implementation.
+The weight-injection pattern is:
+```python
+saved   = torch.load("roste_adapter.pt", map_location="cpu")
+current = dict(model.named_parameters())
+with torch.no_grad():
+    for name, tensor in saved.items():
+        if name in current and current[name].shape == tensor.shape:
+            current[name].copy_(tensor.to(current[name].device, current[name].dtype))
+```

chat_template.jinja ADDED Viewed

	@@ -0,0 +1 @@

+ {%- if messages[0]['role'] == 'system' -%}{{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}{%- else -%}{{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}{%- endif -%}{%- for message in messages -%}{%- if message.role == 'system' and loop.first -%}{# Skip #}{%- else -%}{{- '<|im_start|>' + message['role'] + '\n' + message['content'] + '<|im_end|>' + '\n' }}{%- endif -%}{%- endfor -%}{%- if add_generation_prompt -%}{{- '<|im_start|>assistant\n' }}{%- if enable_thinking is defined and enable_thinking is true -%}{{- '<think>\n' }}{%- endif -%}{%- endif -%}

merges.txt ADDED Viewed

The diff for this file is too large to render. See raw diff

roste_adapter.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:bc37949c86480a0e9538151182c646f0b20b0fd6fac0ab07327046d65b92ab6b
+size 122159843

roste_config.json ADDED Viewed

	@@ -0,0 +1,25 @@

+{
+  "model_name": "ByteDance/Ouro-2.6B-Thinking",
+  "w_bits": 4,
+  "a_bits": 4,
+  "kv_bits": 4,
+  "lora_r": 16,
+  "lora_alpha": 32,
+  "lora_dropout": 0.05,
+  "rotations_used": [
+    "R3 (Q/K head_dim)",
+    "R4 (down_proj input, block-Hadamard)"
+  ],
+  "rotations_skipped": [
+    "R1",
+    "R2"
+  ],
+  "max_seq_len": 1024,
+  "transformers_version": "4.54.1",
+  "trl_version": "0.11.4",
+  "peft_version": "0.13.2",
+  "stages": [
+    "UltraChat-200k",
+    "NVARC-augmented-puzzles"
+  ]
+}

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,49 @@

+{
+  "additional_special_tokens": [
+    "<|endoftext|>",
+    "<|im_start|>",
+    "<|im_end|>",
+    "<think>",
+    "</think>",
+    "<file_sep>",
+    "<filename>",
+    "<gh_stars>",
+    "<issue_start>",
+    "<issue_comment>",
+    "<issue_closed>",
+    "<jupyter_start>",
+    "<jupyter_text>",
+    "<jupyter_code>",
+    "<jupyter_output>",
+    "<jupyter_script>",
+    "<empty_output>"
+  ],
+  "bos_token": {
+    "content": "<|im_start|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "eos_token": {
+    "content": "<|im_end|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": {
+    "content": "<|im_end|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "unk_token": {
+    "content": "<|endoftext|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  }
+}

tokenizer.json ADDED Viewed

The diff for this file is too large to render. See raw diff

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,169 @@

+{
+  "add_prefix_space": false,
+  "added_tokens_decoder": {
+    "0": {
+      "content": "<|endoftext|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "1": {
+      "content": "<|im_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "2": {
+      "content": "<|im_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "3": {
+      "content": "<think>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "4": {
+      "content": "</think>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "5": {
+      "content": "<file_sep>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "6": {
+      "content": "<filename>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "7": {
+      "content": "<gh_stars>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "8": {
+      "content": "<issue_start>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "9": {
+      "content": "<issue_comment>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "10": {
+      "content": "<issue_closed>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "11": {
+      "content": "<jupyter_start>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "12": {
+      "content": "<jupyter_text>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "13": {
+      "content": "<jupyter_code>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "14": {
+      "content": "<jupyter_output>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "15": {
+      "content": "<jupyter_script>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "16": {
+      "content": "<empty_output>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "additional_special_tokens": [
+    "<|endoftext|>",
+    "<|im_start|>",
+    "<|im_end|>",
+    "<think>",
+    "</think>",
+    "<file_sep>",
+    "<filename>",
+    "<gh_stars>",
+    "<issue_start>",
+    "<issue_comment>",
+    "<issue_closed>",
+    "<jupyter_start>",
+    "<jupyter_text>",
+    "<jupyter_code>",
+    "<jupyter_output>",
+    "<jupyter_script>",
+    "<empty_output>"
+  ],
+  "bos_token": "<|im_start|>",
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "<|im_end|>",
+  "extra_special_tokens": {},
+  "model_max_length": 131072,
+  "pad_token": "<|im_end|>",
+  "tokenizer_class": "GPT2Tokenizer",
+  "unk_token": "<|endoftext|>",
+  "vocab_size": 49152
+}

vocab.json ADDED Viewed

The diff for this file is too large to render. See raw diff