Text Generation
Transformers
Safetensors
qwen2
coder
code
agent
conversational
text-generation-inference
Instructions to use AdminReal/NexusCoder with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use AdminReal/NexusCoder with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="AdminReal/NexusCoder") messages = [ {"role": "user", "content": "Who are you?"}, ] pipe(messages)# Load model directly from transformers import AutoTokenizer, AutoModelForCausalLM tokenizer = AutoTokenizer.from_pretrained("AdminReal/NexusCoder") model = AutoModelForCausalLM.from_pretrained("AdminReal/NexusCoder", device_map="auto") messages = [ {"role": "user", "content": "Who are you?"}, ] inputs = tokenizer.apply_chat_template( messages, add_generation_prompt=True, tokenize=True, return_dict=True, return_tensors="pt", ).to(model.device) outputs = model.generate(**inputs, max_new_tokens=40) print(tokenizer.decode(outputs[0][inputs["input_ids"].shape[-1]:])) - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use AdminReal/NexusCoder with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "AdminReal/NexusCoder" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "AdminReal/NexusCoder", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker
docker model run hf.co/AdminReal/NexusCoder
- SGLang
How to use AdminReal/NexusCoder with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "AdminReal/NexusCoder" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "AdminReal/NexusCoder", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "AdminReal/NexusCoder" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "AdminReal/NexusCoder", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }' - Docker Model Runner
How to use AdminReal/NexusCoder with Docker Model Runner:
docker model run hf.co/AdminReal/NexusCoder
Import NexusCoder from github.com/mhieuhonda/NexusCoder
Browse filesThis view is limited to 50 files because it contains too many changes. See raw diff
- .gitignore +74 -0
- .python-version +1 -0
- ADVERTISEMENT.txt +162 -0
- AGENTS.md +118 -0
- ATTRIBUTIONS.md +114 -0
- CHANGELOG.md +349 -0
- CONTRIBUTING.md +66 -0
- LICENSE +193 -0
- README.md +134 -0
- configs/code_corpus.yaml +0 -0
- configs/nexus_coder_10b.yaml +120 -0
- configs/nexus_coder_30b.yaml +103 -0
- configs/nexus_coder_423b.yaml +89 -0
- configs/nexus_coder_70b.yaml +106 -0
- configs/nexus_coder_medium.yaml +80 -0
- configs/nexus_coder_small.yaml +77 -0
- configs/nexus_coder_tiny.yaml +80 -0
- configs/nexus_coder_xlarge.yaml +92 -0
- configs/sources.yaml +685 -0
- data/README.md +9 -0
- docs/ARCHITECTURE.md +95 -0
- docs/DATA.md +188 -0
- docs/SKILLS.md +175 -0
- docs/TOOLS.md +162 -0
- docs/TRAINING.md +77 -0
- nexus/__init__.py +81 -0
- nexus/agent/__init__.py +4 -0
- nexus/agent/agent.py +364 -0
- nexus/agent/memory.py +191 -0
- nexus/agent/planner.py +260 -0
- nexus/agent/router.py +168 -0
- nexus/config.py +569 -0
- nexus/cybergym/__init__.py +100 -0
- nexus/cybergym/adaptive_routing.py +148 -0
- nexus/cybergym/compression.py +150 -0
- nexus/cybergym/context_expansion.py +138 -0
- nexus/cybergym/genome.py +315 -0
- nexus/cybergym/mutation.py +270 -0
- nexus/cybergym/speciation.py +183 -0
- nexus/cybergym/trainer.py +237 -0
- nexus/data/__init__.py +42 -0
- nexus/data/collectors/__init__.py +39 -0
- nexus/data/collectors/arxiv_collector.py +225 -0
- nexus/data/collectors/github_collector.py +420 -0
- nexus/data/collectors/huggingface_collector.py +310 -0
- nexus/data/collectors/python_alpaca_collector.py +117 -0
- nexus/data/collectors/stackoverflow_collector.py +250 -0
- nexus/data/collectors/starcoder2_collector.py +186 -0
- nexus/data/collectors/the_stack_collector.py +131 -0
- nexus/data/collectors/wikipedia_collector.py +164 -0
.gitignore
ADDED
|
@@ -0,0 +1,74 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Byte-compiled / optimized / DLL files
|
| 2 |
+
__pycache__/
|
| 3 |
+
*.py[cod]
|
| 4 |
+
*$py.class
|
| 5 |
+
|
| 6 |
+
# C extensions
|
| 7 |
+
*.so
|
| 8 |
+
|
| 9 |
+
# Distribution / packaging
|
| 10 |
+
.Python
|
| 11 |
+
build/
|
| 12 |
+
develop-eggs/
|
| 13 |
+
dist/
|
| 14 |
+
downloads/
|
| 15 |
+
eggs/
|
| 16 |
+
.eggs/
|
| 17 |
+
lib/
|
| 18 |
+
lib64/
|
| 19 |
+
parts/
|
| 20 |
+
sdist/
|
| 21 |
+
var/
|
| 22 |
+
wheels/
|
| 23 |
+
*.egg-info/
|
| 24 |
+
.installed.cfg
|
| 25 |
+
*.egg
|
| 26 |
+
|
| 27 |
+
# PyInstaller
|
| 28 |
+
*.manifest
|
| 29 |
+
*.spec
|
| 30 |
+
|
| 31 |
+
# Installer logs
|
| 32 |
+
pip-log.txt
|
| 33 |
+
pip-delete-this-directory.txt
|
| 34 |
+
|
| 35 |
+
# Unit test / coverage reports
|
| 36 |
+
htmlcov/
|
| 37 |
+
.tox/
|
| 38 |
+
.coverage
|
| 39 |
+
.coverage.*
|
| 40 |
+
.cache
|
| 41 |
+
nosetests.xml
|
| 42 |
+
coverage.xml
|
| 43 |
+
*.cover
|
| 44 |
+
.pytest_cache/
|
| 45 |
+
|
| 46 |
+
# Jupyter Notebook
|
| 47 |
+
.ipynb_checkpoints
|
| 48 |
+
|
| 49 |
+
# Environments
|
| 50 |
+
.env
|
| 51 |
+
.venv
|
| 52 |
+
env/
|
| 53 |
+
venv/
|
| 54 |
+
ENV/
|
| 55 |
+
|
| 56 |
+
# IDE
|
| 57 |
+
.idea/
|
| 58 |
+
.vscode/
|
| 59 |
+
*.swp
|
| 60 |
+
*.swo
|
| 61 |
+
|
| 62 |
+
# OS
|
| 63 |
+
.DS_Store
|
| 64 |
+
Thumbs.db
|
| 65 |
+
|
| 66 |
+
# Project specific
|
| 67 |
+
checkpoints/
|
| 68 |
+
*.pt
|
| 69 |
+
*.pth
|
| 70 |
+
*.bin
|
| 71 |
+
*.safetensors
|
| 72 |
+
logs/
|
| 73 |
+
*.log
|
| 74 |
+
nexus_coder-*.pt
|
.python-version
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
3.12.13
|
ADVERTISEMENT.txt
ADDED
|
@@ -0,0 +1,162 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
============================================================================
|
| 2 |
+
NEXUS CODER v0.4 — CYBERFORGE EDITION
|
| 3 |
+
AI Code & Security Engine — Open Source
|
| 4 |
+
============================================================================
|
| 5 |
+
|
| 6 |
+
Created by: Hieu Louis
|
| 7 |
+
GitHub: https://github.com/mhieuhonda/NexusCoder
|
| 8 |
+
Year: 2026
|
| 9 |
+
License: NAL-1.0 (Attribution Required)
|
| 10 |
+
Version: 0.4.0
|
| 11 |
+
Python: 3.12.13
|
| 12 |
+
|
| 13 |
+
|
| 14 |
+
┌──────────────────────────────────────────────────────────────────────────┐
|
| 15 |
+
│ │
|
| 16 |
+
│ NEXUS CODER — SIÊU AI MÃ NGUỒN MỞ CHO CODE & BẢO MẬT │
|
| 17 |
+
│ │
|
| 18 |
+
│ • 423 tỷ tham số tổng, 39 tỷ tham số kích hoạt mỗi token │
|
| 19 |
+
│ • Cửa sổ ngữ cảnh 3 TRIỆU tokens │
|
| 20 |
+
│ • 60+ kỹ năng (skills) tích hợp │
|
| 21 |
+
│ • 80+ công cụ (tools) tự động đăng ký │
|
| 22 |
+
│ • Kiến trúc MoE Transformer thế hệ mới │
|
| 23 |
+
│ │
|
| 24 |
+
└──────────────────────────────────────────────────────────────────────────┘
|
| 25 |
+
|
| 26 |
+
|
| 27 |
+
TẠI SAO NEXUS CODER KHÁC BIỆT?
|
| 28 |
+
==============================
|
| 29 |
+
|
| 30 |
+
Nexus Coder v0.4 là một kiến trúc AI mã nguồn mở hoàn chỉnh, được Hieu Louis
|
| 31 |
+
thiết kế từ con số không. Repository này cung cấp:
|
| 32 |
+
|
| 33 |
+
✓ Toàn bộ mã nguồn kiến trúc model (Python/PyTorch)
|
| 34 |
+
✓ Pipeline thu thập và xử lý dữ liệu code từ hàng nghìn GitHub repos
|
| 35 |
+
✓ Framework huấn luyện đa giai đoạn
|
| 36 |
+
✓ 60+ skills (code generation, debugging, security audit, ...)
|
| 37 |
+
✓ 80+ tools (file ops, exec, web, database, devops, ...)
|
| 38 |
+
✓ Tương thích Python 3.12.13 (strict)
|
| 39 |
+
|
| 40 |
+
|
| 41 |
+
TRUNG THỰC VỀ TRẠNG THÁI MODEL
|
| 42 |
+
================================
|
| 43 |
+
|
| 44 |
+
⚠ REPO NÀY KHÔNG CHỨA MODEL ĐÃ ĐƯỢC TRAIN.
|
| 45 |
+
|
| 46 |
+
Nexus Coder v0.4 phân phối MÃ NGUỒN của kiến trúc, pipeline dữ liệu,
|
| 47 |
+
và framework huấn luyện. Người dùng tự huấn luyện mô hình trên dữ
|
| 48 |
+
liệu của mình. Mọi thông tin quảng cáo về "performance" hay "benchmark"
|
| 49 |
+
chỉ là ước tính lý thuyết dựa trên kích thước kiến trúc — chưa có
|
| 50 |
+
model thực tế nào được train và đánh giá chính thức.
|
| 51 |
+
|
| 52 |
+
Khi bạn thấy ai đó chia sẻ "Nexus Coder đã đạt X điểm benchmark Y", hãy
|
| 53 |
+
hỏi xem họ có train model thực tế hay không, và với dữ liệu gì.
|
| 54 |
+
|
| 55 |
+
|
| 56 |
+
TÍNH NĂNG KỸ THUẬT CHÍNH
|
| 57 |
+
==========================
|
| 58 |
+
|
| 59 |
+
• Kiến trúc MoE Transformer với GQA (Grouped Query Attention)
|
| 60 |
+
• RoPE + YaRN scaling cho context window cực dài (3M tokens)
|
| 61 |
+
• FlashAttention-2 + SDPA + manual fallback
|
| 62 |
+
• Sliding Window Attention cho long-context efficiency
|
| 63 |
+
• QK-norm (Llama-3 style) cho training stability
|
| 64 |
+
• KV cache quantization (int8 / fp8) cho inference memory
|
| 65 |
+
• MLP-parallel (gate + up fuses thành 1 matmul)
|
| 66 |
+
• Gradient checkpointing cho training VRAM tiết kiệm
|
| 67 |
+
• Adaptive Density Routing (top-2 → top-8 experts theo input)
|
| 68 |
+
• 48 experts chuyên biệt hóa theo domain code (Python, JS, Rust, ...)
|
| 69 |
+
|
| 70 |
+
• 8 nguồn dữ liệu: GitHub curated corpus (1000+ repos), HuggingFace,
|
| 71 |
+
arXiv, Wikipedia, StackOverflow, The-Stack v2, StarCoder2-data,
|
| 72 |
+
Python-Alpaca
|
| 73 |
+
|
| 74 |
+
• Tích hợp 5 framework tham chiếu: litgpt, LlamaFactory, axolotl,
|
| 75 |
+
OpenHands, omp-gym (xem ATTRIBUTIONS.md)
|
| 76 |
+
|
| 77 |
+
|
| 78 |
+
CẤU HÌNH VARIANTS
|
| 79 |
+
=================
|
| 80 |
+
|
| 81 |
+
tiny — 5M params (CPU demo)
|
| 82 |
+
small — 125M params (1 GPU)
|
| 83 |
+
medium — 1B params (4-8 GPU)
|
| 84 |
+
large — 10B params (32+ GPU) — backward-compat với v0.3
|
| 85 |
+
xlarge — ~30B params (64+ GPU)
|
| 86 |
+
30b — 30B/3B (64-128 GPU, H100 cluster)
|
| 87 |
+
70b — ~70B/~12B (research only)
|
| 88 |
+
423b — 423B/39B + 3M context (DEFAULT v0.4) — frontier scale
|
| 89 |
+
|
| 90 |
+
|
| 91 |
+
CÀI ĐẶT
|
| 92 |
+
========
|
| 93 |
+
|
| 94 |
+
git clone https://github.com/mhieuhonda/NexusCoder.git
|
| 95 |
+
cd NexusCoder
|
| 96 |
+
python3.12.13 -m venv venv
|
| 97 |
+
source venv/bin/activate
|
| 98 |
+
pip install -r requirements.txt
|
| 99 |
+
|
| 100 |
+
|
| 101 |
+
SỬ DỤNG
|
| 102 |
+
========
|
| 103 |
+
|
| 104 |
+
# Xem tóm tắt cấu hình
|
| 105 |
+
python -c "from nexus.config import print_config_summary; print_config_summary()"
|
| 106 |
+
|
| 107 |
+
# Tiny demo
|
| 108 |
+
python scripts/train.py --config tiny --steps 100
|
| 109 |
+
|
| 110 |
+
# Train (cần GPU)
|
| 111 |
+
python scripts/train.py --config large --steps 5000 --use-amp
|
| 112 |
+
|
| 113 |
+
# Chat với Nexus Agent
|
| 114 |
+
python scripts/chat.py
|
| 115 |
+
|
| 116 |
+
|
| 117 |
+
GIẤY PHÉP — NAL-1.0 (ATTRIBUTION REQUIRED)
|
| 118 |
+
==========================================
|
| 119 |
+
|
| 120 |
+
Nexus Coder v0.4 được phát hành dưới giấy phép NexusCoder Attribution
|
| 121 |
+
License v1.0 (NAL-1.0). Bạn được phép:
|
| 122 |
+
|
| 123 |
+
✓ Sử dụng cho bất kỳ mục đích nào (commercial hoặc non-commercial)
|
| 124 |
+
✓ Sửa đổi, phân phối, sublicense
|
| 125 |
+
✓ Train, fine-tune, distill, quantize, ...
|
| 126 |
+
✓ Build sản phẩm, dịch vụ, nghiên cứu trên nền Nexus Coder
|
| 127 |
+
|
| 128 |
+
BẮT BUỘC:
|
| 129 |
+
|
| 130 |
+
• Phải ghi danh tác giả gốc: "Hieu Louis"
|
| 131 |
+
• Phải kèm link: https://github.com/mhieuhonda/NexusCoder
|
| 132 |
+
• Trong model cards, README, UI, About pages, API responses,
|
| 133 |
+
research citations — bất cứ nơi nào hợp lý và thông dụng.
|
| 134 |
+
|
| 135 |
+
Không được:
|
| 136 |
+
✗ Xóa hoặc làm mờ attribution notices
|
| 137 |
+
✗ Cầm quyền tác giả của người khác
|
| 138 |
+
✗ Implement technical measures để erase embedded authorship
|
| 139 |
+
|
| 140 |
+
Xem LICENSE để biết chi tiết đầy đủ.
|
| 141 |
+
|
| 142 |
+
|
| 143 |
+
TÁC GIẢ
|
| 144 |
+
========
|
| 145 |
+
|
| 146 |
+
Hieu Louis — 2026
|
| 147 |
+
GitHub: https://github.com/mhieuhonda
|
| 148 |
+
Project: https://github.com/mhieuhonda/NexusCoder
|
| 149 |
+
License: NAL-1.0 (Attribution Required)
|
| 150 |
+
|
| 151 |
+
|
| 152 |
+
KẾT LUẬN
|
| 153 |
+
========
|
| 154 |
+
|
| 155 |
+
Nexus Coder v0.4 là một dự án mã nguồn mở cá nhân, được xây dựng với
|
| 156 |
+
tinh thần minh bạch: đây là CODE của kiến trúc AI, KHÔNG PHẢI model đã
|
| 157 |
+
train. Bạn có thể dùng nó để tự train model của mình, học hỏi kiến trúc,
|
| 158 |
+
hoặc build sản phẩm riêng. Hãy ghi danh Hieu Louis khi làm vậy.
|
| 159 |
+
|
| 160 |
+
Cảm ơn bạn đã quan tâm đến Nexus Coder.
|
| 161 |
+
|
| 162 |
+
— Hieu Louis, 2026
|
AGENTS.md
ADDED
|
@@ -0,0 +1,118 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# AGENTS.md — Guide for AI Agents Contributing to Nexus Coder
|
| 2 |
+
|
| 3 |
+
> **Read this first.** This file is written for AI agents who want to contribute
|
| 4 |
+
> to this repository. Humans can skip to `CONTRIBUTING.md`.
|
| 5 |
+
|
| 6 |
+
## What Nexus Coder Is (and Is Not)
|
| 7 |
+
|
| 8 |
+
Nexus Coder is an **open AI architecture** — source code, data pipeline, and a
|
| 9 |
+
training framework for a large Mixture-of-Experts (MoE) Transformer focused on
|
| 10 |
+
code generation and security analysis.
|
| 11 |
+
|
| 12 |
+
**Important honesty note:** the model is **NOT pretrained**. There are no
|
| 13 |
+
checkpoints, no weights, no benchmarks. What exists is the architecture source
|
| 14 |
+
(60+ skills, 80+ tools, model layers, tokenizer, trainers, data collectors).
|
| 15 |
+
Your contribution helps turn this architecture into something that can actually
|
| 16 |
+
run and learn.
|
| 17 |
+
|
| 18 |
+
If you expected a finished model, stop here. If you want to help build one from
|
| 19 |
+
the ground up, you are in the right place.
|
| 20 |
+
|
| 21 |
+
## How to Contribute (Agent Workflow)
|
| 22 |
+
|
| 23 |
+
### 1. Understand the layout
|
| 24 |
+
|
| 25 |
+
```
|
| 26 |
+
nexus/
|
| 27 |
+
├── model/ # MoE Transformer layers (attention, MoE, RoPE, ...)
|
| 28 |
+
├── tokenizer/ # tokenizer implementation
|
| 29 |
+
├── training/ # trainer + dataset pipeline
|
| 30 |
+
├── inference/ # generation
|
| 31 |
+
├── agent/ # planner, router, memory, safety
|
| 32 |
+
├── skills/ # 60+ skills (auto-discovery)
|
| 33 |
+
├── tools/ # 80+ tools (auto-discovery)
|
| 34 |
+
├── data/ # collectors + processors
|
| 35 |
+
├── optim/ # quantization, LoRA, distillation, pruning
|
| 36 |
+
├── safety/ # filters, guardrails
|
| 37 |
+
├── eval/ # benchmarks, metrics
|
| 38 |
+
└── integrations/ # litgpt, LlamaFactory, axolotl, OpenHands, omp-gym
|
| 39 |
+
```
|
| 40 |
+
|
| 41 |
+
### 2. Pick an entry point
|
| 42 |
+
|
| 43 |
+
Good first tasks for agents:
|
| 44 |
+
|
| 45 |
+
- **Run the tiny demo on CPU**: `python scripts/train.py --config tiny --steps 100`
|
| 46 |
+
— verify it works, report output.
|
| 47 |
+
- **Test a skill**: import a skill from `nexus/skills/` and exercise it against
|
| 48 |
+
sample input. Report what works / what breaks.
|
| 49 |
+
- **Test a tool**: pick a tool in `nexus/tools/` and exercise it.
|
| 50 |
+
- **Review an integration**: read `nexus/integrations/*.py` and report whether
|
| 51 |
+
the adapted pattern is faithful to the upstream project.
|
| 52 |
+
- **Improve docs**: `docs/` always needs sharper explanations.
|
| 53 |
+
- **Add tests**: `tests/` is thin — any new test is valuable.
|
| 54 |
+
|
| 55 |
+
### 3. Workflow
|
| 56 |
+
|
| 57 |
+
1. **Open an issue first** describing what you want to do (unless it is a
|
| 58 |
+
trivial fix). This prevents wasted work.
|
| 59 |
+
2. **Fork** the repo (or ask a maintainer for collaborator access).
|
| 60 |
+
3. Create a **branch**, make your change, and open a **pull request**.
|
| 61 |
+
4. In your PR description, state clearly:
|
| 62 |
+
- what you changed,
|
| 63 |
+
- how you verified it (command + output),
|
| 64 |
+
- which config/skill/tool you exercised.
|
| 65 |
+
|
| 66 |
+
### 4. Self-sufficiency
|
| 67 |
+
|
| 68 |
+
You are expected to **figure things out yourself**:
|
| 69 |
+
|
| 70 |
+
- Setup: `python3.12.13 -m venv venv && pip install -r requirements.txt`
|
| 71 |
+
- Small configs (`tiny`, `small`) run on **CPU** — no GPU needed.
|
| 72 |
+
- Larger configs need real hardware; if you have GPUs, train and report.
|
| 73 |
+
- If something is ambiguous, open an issue rather than guessing silently.
|
| 74 |
+
|
| 75 |
+
## You May Train the Model — Freely
|
| 76 |
+
|
| 77 |
+
You are **explicitly welcome and encouraged to train the model yourself**. The
|
| 78 |
+
NAL-1.0 license allows it for any purpose. There is no gate, no permission
|
| 79 |
+
request needed, and no fee.
|
| 80 |
+
|
| 81 |
+
- **Small configs** (`tiny`, `small`) run on CPU — try a real training run
|
| 82 |
+
today with zero GPU cost.
|
| 83 |
+
- **Larger configs** (`medium`, `large`, `423b`) need serious hardware. If you
|
| 84 |
+
have GPUs (or a cloud budget), you are free to use them.
|
| 85 |
+
- Use the bundled data pipeline (`nexus/data/`) or your own data — both are fine.
|
| 86 |
+
- **Report your results honestly.** If you train something and get real
|
| 87 |
+
numbers, open a PR or an issue with your findings. The community (and the
|
| 88 |
+
project) wants to see real training logs, not hype.
|
| 89 |
+
|
| 90 |
+
The project does not provide compute. You bring the hardware; we provide the
|
| 91 |
+
architecture, the data pipeline, and the training framework.
|
| 92 |
+
|
| 93 |
+
> **Note for fairness:** training a model on this architecture does not make
|
| 94 |
+
> you a contributor to the repository. Attribution requirements still apply
|
| 95 |
+
> (NAL-1.0) whenever you use the trained result. To be listed as a contributor,
|
| 96 |
+
> submit a PR that improves the repo.
|
| 97 |
+
|
| 98 |
+
## Attribution & License
|
| 99 |
+
|
| 100 |
+
- License: **NAL-1.0** (Attribution Required). You may use/modify/distribute/
|
| 101 |
+
train freely, but must credit the original author **Hieu Louis**
|
| 102 |
+
(github.com/mhieuhonda) wherever the work is used.
|
| 103 |
+
- Contributors are listed in `ATTRIBUTIONS.md`. If your PR merges, add yourself
|
| 104 |
+
there (or ask a maintainer to add you).
|
| 105 |
+
|
| 106 |
+
## Communication
|
| 107 |
+
|
| 108 |
+
- GitHub Issues + PRs are the coordination surface.
|
| 109 |
+
- For discussion, join the **code-realm** community on Moltbook (ask the
|
| 110 |
+
maintainer for the invite). The project has an active presence there.
|
| 111 |
+
|
| 112 |
+
## Golden Rules
|
| 113 |
+
|
| 114 |
+
1. **Be honest** — this is an untrained architecture. Never claim benchmark
|
| 115 |
+
results that do not exist.
|
| 116 |
+
2. **Small, verifiable PRs** beat big unverifiable ones.
|
| 117 |
+
3. **Reproduce before you report** — always run the thing you are claiming.
|
| 118 |
+
4. **Credit the author** in any downstream work (NAL-1.0).
|
ATTRIBUTIONS.md
ADDED
|
@@ -0,0 +1,114 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Attributions
|
| 2 |
+
|
| 3 |
+
Nexus Coder v0.3 adapts ideas and code patterns from the following open-source projects.
|
| 4 |
+
All credit for the original algorithms goes to their respective authors. The code in
|
| 5 |
+
`nexus/integrations/` is rewritten to integrate cleanly into Nexus Coder's architecture;
|
| 6 |
+
it is NOT a vendored copy.
|
| 7 |
+
|
| 8 |
+
## Reference Frameworks
|
| 9 |
+
|
| 10 |
+
### 1. LitGPT (Lightning AI)
|
| 11 |
+
- **License**: Apache 2.0
|
| 12 |
+
- **Source**: https://github.com/Lightning-AI/litgpt
|
| 13 |
+
- **What we adapted**:
|
| 14 |
+
- RoPE scaling strategies (linear / NTK-aware / YaRN) → `nexus/model/rope.py`
|
| 15 |
+
- FusedLinear pattern (concatenated Q/K/V projections) → `nexus/integrations/litgpt.py`
|
| 16 |
+
- PyTorch SDPA backend selection → `nexus/model/flash_attention.py`
|
| 17 |
+
- **Original attribution**: LitGPT: Lightning AI's LLM training toolkit. Authors: Karpathy et al. (Lightning AI), 2023-2024.
|
| 18 |
+
|
| 19 |
+
### 2. LLaMA Factory (hiyouga)
|
| 20 |
+
- **License**: Apache 2.0
|
| 21 |
+
- **Source**: https://github.com/hiyouga/LlamaFactory (also https://github.com/hiyouga/LLaMA-Factory)
|
| 22 |
+
- **What we adapted**:
|
| 23 |
+
- Dataset format converters (Alpaca / ShareGPT / ChatML / Completion → unified Nexus format) → `nexus/integrations/llamafactory.py`
|
| 24 |
+
- Concept of unified dataset registry → `nexus/data/collectors/`
|
| 25 |
+
- **Original attribution**: LlamaFactory: Unify Fine-tuning 100+ LLMs. Author: hiyouga.
|
| 26 |
+
|
| 27 |
+
### 3. Axolotl (axolotl-ai-cloud)
|
| 28 |
+
- **License**: Apache 2.0
|
| 29 |
+
- **Source**: https://github.com/axolotl-ai-cloud/axolotl
|
| 30 |
+
- **What we adapted**:
|
| 31 |
+
- AxolotlStyleConfig dataclass (typed training config schema) → `nexus/integrations/axolotl.py`
|
| 32 |
+
- Concept of single-YAML training configuration
|
| 33 |
+
- **Original attribution**: Axolotl: a simple tool for fine-tuning LLMs. Authors: winglian + axolotl-ai-cloud contributors.
|
| 34 |
+
|
| 35 |
+
### 4. OpenHands
|
| 36 |
+
- **License**: MIT
|
| 37 |
+
- **Source**: https://github.com/OpenHands/OpenHands
|
| 38 |
+
- **What we adapted**:
|
| 39 |
+
- AgentLoop pattern (planner / executor / observer / reflector) → `nexus/integrations/openhands.py`
|
| 40 |
+
- Concept of structured agent loop with reflection
|
| 41 |
+
- **Original attribution**: OpenHands (formerly OpenDevin): an open platform for AI software developers. Authors: OpenHands contributors.
|
| 42 |
+
|
| 43 |
+
### 5. omp-gym (Dylan Tirandaz)
|
| 44 |
+
- **License**: MIT
|
| 45 |
+
- **Source**: https://github.com/dylantirandaz/omp-gym
|
| 46 |
+
- **What we adapted**:
|
| 47 |
+
- OpenMP optimization benchmark tasks → `nexus/integrations/omp_gym.py`
|
| 48 |
+
- Concept of "predict-the-optimization" eval task
|
| 49 |
+
- **Original attribution**: omp-gym: An OpenMP optimization gym environment. Author: Dylan Tirandaz.
|
| 50 |
+
|
| 51 |
+
## Other Attribution
|
| 52 |
+
|
| 53 |
+
### Algorithms implemented in `nexus/model/`
|
| 54 |
+
- **RoPE**: Su et al., "RoFormer: Enhanced Transformer with Rotary Position Embedding" (2021). https://arxiv.org/abs/2104.09864
|
| 55 |
+
- **FlashAttention**: Dao et al., "FlashAttention: Fast and Memory-Efficient Exact Attention with IO-Awareness" (2022). https://arxiv.org/abs/2205.14135
|
| 56 |
+
- **FlashAttention-2**: Dao, "FlashAttention-2: Faster Attention with Better Parallelism and Work Partitioning" (2023). https://arxiv.org/abs/2307.08691
|
| 57 |
+
- **ALiBi**: Press et al., "Train Short, Test Long: Attention with Linear Biases Enables Input Length Extrapolation" (ICLR 2022). https://arxiv.org/abs/2108.12409
|
| 58 |
+
- **Sliding Window Attention**: Beltagy et al., "Longformer: The Long-Document Transformer" (2020). https://arxiv.org/abs/2004.05150
|
| 59 |
+
- **YaRN**: Peng et al., "YaRN: Efficient Context Window Extension of Large Language Models" (2023). https://arxiv.org/abs/2309.00071
|
| 60 |
+
- **NTK-aware RoPE scaling**: bloc97, "NTK-Aware Scaled RoPE" (2023). https://www.reddit.com/r/LocalLLaMA/comments/14lzrgj/
|
| 61 |
+
- **SwiGLU**: Shazeer, "GLU Variants Improve Transformer" (2020). https://arxiv.org/abs/2002.05202
|
| 62 |
+
- **RMSNorm**: Zhang & Sennrich, "Root Mean Square Layer Normalization" (2019). https://arxiv.org/abs/1910.07467
|
| 63 |
+
- **GQA**: Ainslie et al., "GQA: Training Generalized Multi-Query Transformer Models from Multi-Head Checkpoints" (2023). https://arxiv.org/abs/2305.13245
|
| 64 |
+
- **MoE**: Shazeer et al., "Outrageously Large Neural Networks: The Sparsely-Gated Mixture-of-Experts Layer" (2017). https://arxiv.org/abs/1701.06538
|
| 65 |
+
- **Switch Transformer**: Fedus et al., "Switch Transformers: Scaling to Trillion Parameter Models with Simple and Efficient Sparsity" (2021). https://arxiv.org/abs/2101.03961
|
| 66 |
+
|
| 67 |
+
### Datasets referenced in `configs/sources.yaml`
|
| 68 |
+
- **The-Stack v2**: BigCode, https://huggingface.co/datasets/bigcode/the-stack-v2-train-full-ids
|
| 69 |
+
- **StarCoder2-data**: BigCode, https://huggingface.co/datasets/bigcode/starcoder2data
|
| 70 |
+
- **CodeParrot**: CodeParrot, https://huggingface.co/codeparrot
|
| 71 |
+
- **Wikipedia**: Wikimedia, https://huggingface.co/wikimedia/wikipedia
|
| 72 |
+
- **OSCAR**: https://oscar-project.org
|
| 73 |
+
- **UltraChat**: HuggingFaceH4, https://huggingface.co/HuggingFaceH4/ultrachat_200k
|
| 74 |
+
- **OpenHermes**: teknium, https://huggingface.co/teknium/OpenHermes-2.5
|
| 75 |
+
- **OpenOrca**: https://huggingface.co/Open-Orca/OpenOrca
|
| 76 |
+
- **MetaMathQA**: https://huggingface.co/meta-math/MetaMathQA
|
| 77 |
+
- **GSM8K**: https://huggingface.co/datasets/gsm8k
|
| 78 |
+
- **HumanEval**: OpenAI, https://huggingface.co/datasets/openai_humaneval
|
| 79 |
+
- **MBPP**: Google Research, https://huggingface.co/datasets/mbpp
|
| 80 |
+
- **MATH**: https://huggingface.co/datasets/competition_math
|
| 81 |
+
- **FineWeb**: HuggingFaceFW, https://huggingface.co/datasets/HuggingFaceFW/fineweb
|
| 82 |
+
- **Open-Web-Math**: https://huggingface.co/datasets/open-web-math/open-web-math
|
| 83 |
+
- **Dolma**: AllenAI, https://huggingface.co/datasets/allenai/dolma
|
| 84 |
+
- **Pile**: EleutherAI, https://huggingface.co/datasets/EleutherAI/pile
|
| 85 |
+
- **C4**: Google, https://huggingface.co/datasets/c4
|
| 86 |
+
|
| 87 |
+
### Tools inspired by existing libraries
|
| 88 |
+
- The `Tool` and `Skill` base classes follow the OpenAI function-calling schema pattern
|
| 89 |
+
- Database tools wrap established client libraries (psycopg2, pymysql, redis, pymongo, etc.)
|
| 90 |
+
- Web tools use `requests` + `BeautifulSoup` conventions
|
| 91 |
+
|
| 92 |
+
## License
|
| 93 |
+
|
| 94 |
+
Nexus Coder is licensed under the MIT License (see [LICENSE](LICENSE)).
|
| 95 |
+
|
| 96 |
+
The adaptations from the above projects comply with their respective licenses:
|
| 97 |
+
- Apache 2.0 components: retain notice, state changes
|
| 98 |
+
- MIT components: retain copyright notice
|
| 99 |
+
|
| 100 |
+
Where algorithms are reimplemented from academic papers, the original papers
|
| 101 |
+
are cited in the source files.
|
| 102 |
+
|
| 103 |
+
---
|
| 104 |
+
|
| 105 |
+
*This file is part of Nexus Coder v0.3 by Hieu Louis (2026).*
|
| 106 |
+
|
| 107 |
+
|
| 108 |
+
## Contributors
|
| 109 |
+
|
| 110 |
+
> Maintained by hand. Add yourself here when your PR is merged, or ask a
|
| 111 |
+
> maintainer to add you. AI agents are welcome contributors.
|
| 112 |
+
|
| 113 |
+
| Date | Contributor | Contribution |
|
| 114 |
+
|------|-------------|--------------|
|
CHANGELOG.md
ADDED
|
@@ -0,0 +1,349 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Thay đổi / Changelog
|
| 2 |
+
|
| 3 |
+
## v0.4.0 - 2026-08-17 — CyberForge Edition
|
| 4 |
+
|
| 5 |
+
### SUPREME UPGRADE — 423B params, 3M context, CyberGym training methodology
|
| 6 |
+
|
| 7 |
+
**Tác giả / Author**: Hieu Louis
|
| 8 |
+
|
| 9 |
+
#### New Features
|
| 10 |
+
|
| 11 |
+
##### Model architecture — 423B / 39B / 3M context
|
| 12 |
+
- New config `423b` (DEFAULT for v0.4): 423B total / 39B active params
|
| 13 |
+
- 24 layers, hidden 7168, 48 experts (4 active), inter 16384
|
| 14 |
+
- 3,000,000-token context window via YaRN RoPE scaling (×60)
|
| 15 |
+
- Sliding window 32k + QK-norm + KV cache int8 + gradient checkpointing
|
| 16 |
+
- Adaptive Density Routing: top-2 → top-8 active experts based on input entropy
|
| 17 |
+
|
| 18 |
+
##### CyberGym training methodology (NEW)
|
| 19 |
+
- **Code Genome Initialization (CGI)**: weight init from code motifs
|
| 20 |
+
- **Mutation Pressure Training (MPT)**: beneficial weight perturbations during training
|
| 21 |
+
- **Expert Speciation Curriculum (ESC)**: 48 experts → 48 species (Python/JS/Rust/Go/...)
|
| 22 |
+
- **Recursive Self-Compression (RSC)**: periodic self-distillation snapshots
|
| 23 |
+
- **Context Expansion Protocol (CEP)**: progressive 32k → 3M context extension
|
| 24 |
+
- **Adaptive Density Routing (ADR)**: entropy-based top-k routing
|
| 25 |
+
- Orchestrator `CyberForgeTrainer` wires all components together
|
| 26 |
+
|
| 27 |
+
##### Data pipeline — Code corpus curated
|
| 28 |
+
- `configs/code_corpus.yaml`: 1000+ curated GitHub repos across 17 categories
|
| 29 |
+
- Categories: python_core, python_web, python_data, python_ml, python_dl,
|
| 30 |
+
python_tools, javascript_core, javascript_frameworks, rust_core, go_core,
|
| 31 |
+
java_core, c_cpp, devops, security, ai_tools, scientific, systems
|
| 32 |
+
|
| 33 |
+
#### Bug Fixes (48 total)
|
| 34 |
+
|
| 35 |
+
##### CRITICAL (6 fixes)
|
| 36 |
+
- `nexus/safety/__init__.py`: missing `get_default_guardrails` export broke `nexus.agent`
|
| 37 |
+
- `nexus/data/processors/deduplicator.py`: wrong import path (`.._logging_helpers` → `...utils.logging`)
|
| 38 |
+
- `nexus/model/attention.py`: INT8 KV cache quantization discarded scale → crash on 2nd decode step
|
| 39 |
+
- `scripts/collect_data.py`: `CURATED_TAGS` was a class attribute, not module-level → ImportError
|
| 40 |
+
- `nexus/agent/planner.py`: invalid dependency IDs silently treated as "met" (security bug)
|
| 41 |
+
- `nexus/config.py`: 30B / 70B configs were 5×–9× off their advertised size
|
| 42 |
+
|
| 43 |
+
##### MAJOR (22 fixes)
|
| 44 |
+
- MoE never received `attention_mask` (padded tokens polluted aux loss)
|
| 45 |
+
- LoRA `target_modules` listed `gate_proj`/`up_proj` but v0.3 SwiGLU fuses them into `gate_up_proj`
|
| 46 |
+
- `python_exec` sandbox: when run as script, `__builtins__` was a module → sandbox escape
|
| 47 |
+
- `python_exec`: timeout was computed but never enforced → infinite loops could hang the agent
|
| 48 |
+
- `shell.py`: dead `if False` branch with unimported `os`
|
| 49 |
+
- ALiBi `max_slope` parameter was hardcoded to 8.0 (parameter had no effect)
|
| 50 |
+
- ALiBi non-power-of-2 head count subselection was wrong (took first N, not closest N)
|
| 51 |
+
- GitHub collector: `"c++"` language key didn't exist in EXTENSIONS (should be `"cpp"`)
|
| 52 |
+
- GitHub collector: hardcoded `--branch main` failed for repos using `master`
|
| 53 |
+
- arXiv collector: `.find().text` without None check crashed entire parse on missing element
|
| 54 |
+
- arXiv collector: query string not URL-encoded
|
| 55 |
+
- `compute_rouge`: rouge_1 was precision, not recall (corrected to F1)
|
| 56 |
+
- `compute_bleu`: empty references list crashed `min()` call
|
| 57 |
+
- FP8 quantization skip_layers comparison never matched (all params got FP8-quantized)
|
| 58 |
+
- Attention mask shape mismatch with KV cache + sliding window
|
| 59 |
+
- Trainer: AMP scaler state not checkpointed (resume caused NaN gradients)
|
| 60 |
+
- `quality_filter`: off-by-one in 10-gram repetition window
|
| 61 |
+
- `dataset.py`: hardcoded pad id 0 (collided with token 0 if user changed `pad_token_id`)
|
| 62 |
+
- Tokenizer: Vietnamese char `Ẵ` was duplicated as `Ẳ` (missing `Ẵ`)
|
| 63 |
+
- Tokenizer: BPE merge lost `</w>` marker when first symbol had it
|
| 64 |
+
- Tokenizer: `tuple(k.split("|"))` broke when token contained `|`
|
| 65 |
+
- `scripts/train.py`: `--config` choices missing `30b`, `70b`, `423b`
|
| 66 |
+
|
| 67 |
+
##### MINOR (20 fixes)
|
| 68 |
+
- Various unused imports, dead code, type hints
|
| 69 |
+
- See git log for full list
|
| 70 |
+
|
| 71 |
+
#### License change
|
| 72 |
+
- Switched from MIT to **NexusCoder Attribution License v1.0 (NAL-1.0)**
|
| 73 |
+
- Free use for any purpose (commercial/non-commercial/research)
|
| 74 |
+
- Mandatory attribution: "Hieu Louis" + link to original repo
|
| 75 |
+
- See [LICENSE](LICENSE) for full terms
|
| 76 |
+
|
| 77 |
+
#### Files added
|
| 78 |
+
- `nexus/cybergym/__init__.py`
|
| 79 |
+
- `nexus/cybergym/mutation.py`
|
| 80 |
+
- `nexus/cybergym/genome.py`
|
| 81 |
+
- `nexus/cybergym/adaptive_routing.py`
|
| 82 |
+
- `nexus/cybergym/speciation.py`
|
| 83 |
+
- `nexus/cybergym/compression.py`
|
| 84 |
+
- `nexus/cybergym/context_expansion.py`
|
| 85 |
+
- `nexus/cybergym/trainer.py`
|
| 86 |
+
- `configs/nexus_coder_423b.yaml`
|
| 87 |
+
- `configs/code_corpus.yaml`
|
| 88 |
+
- `ADVERTISEMENT.txt`
|
| 89 |
+
|
| 90 |
+
---
|
| 91 |
+
|
| 92 |
+
## v0.3.0 - 2026-08-16
|
| 93 |
+
|
| 94 |
+
### 🚀 MASSIVE UPGRADE - Architecture + 4× Skills + 4× Tools + Massive Data
|
| 95 |
+
|
| 96 |
+
**Tác giả / Author**: Hieu Louis
|
| 97 |
+
|
| 98 |
+
#### ✨ Tính năng mới / New Features
|
| 99 |
+
|
| 100 |
+
##### 🏗️ Kiến trúc v0.3 (NEW)
|
| 101 |
+
- ✅ **FlashAttention-2**: Optional `flash_attn` package backend (falls back to SDPA)
|
| 102 |
+
- ✅ **ALiBi position bias**: Alternative to RoPE for long-context extrapolation (Press et al., 2022)
|
| 103 |
+
- ✅ **Sliding Window Attention**: Alternating SWA / global layers (Longformer / Mistral style)
|
| 104 |
+
- ✅ **QK-norm**: RMSNorm on query/key for training stability (Llama-3 style)
|
| 105 |
+
- ✅ **MLP-parallel**: Fused gate+up projection (concatenated matmul) — faster on modern GPUs
|
| 106 |
+
- ✅ **KV cache quantization**: int8 / fp8 options for inference memory reduction
|
| 107 |
+
- ✅ **Gradient checkpointing**: Trade compute for VRAM at training time
|
| 108 |
+
- ✅ **RoPE scaling strategies**: linear / dynamic (NTK) / ntk / yarn — supports context extension up to 256k
|
| 109 |
+
|
| 110 |
+
##### 📊 Multi-Variant Configs (7 variants)
|
| 111 |
+
- ✅ `tiny` - ~5M params (CPU demo)
|
| 112 |
+
- ✅ `small` - ~125M params (1 GPU)
|
| 113 |
+
- ✅ `medium` - ~1B params (4-8 GPU)
|
| 114 |
+
- ✅ `large` - 10B/1.5B (default, 32+ GPU)
|
| 115 |
+
- ✅ `xlarge` - ~30B/3B (research, 64+ GPU)
|
| 116 |
+
- ✅ `30b` - 30B/3B (v0.3 NEW, 64-128 H100, 64k context)
|
| 117 |
+
- ✅ `70b` - 70B/5B (v0.3 NEW, 256+ H100/H200, 128k context with YaRN ×4)
|
| 118 |
+
|
| 119 |
+
##### 🎯 Skills System (15 → 60+)
|
| 120 |
+
- ✅ **Existing 15**: code_generation, code_review, code_refactor, debugging, documentation, testing, algorithm_design, data_analysis, translation, summarization, reasoning, math_skill, sql_generation, security_audit, performance_opt
|
| 121 |
+
- ✅ **DevOps (5 NEW)**: devops_skill, ci_cd_pipeline, release_management, monitoring, logging_analytics
|
| 122 |
+
- ✅ **ML (10 NEW)**: ml_training, ml_inference, ml_evaluation, ml_data_preprocessing, ml_feature_engineering, ml_hyperparameter_tuning, ml_model_explainability, ml_model_selection, ml_metrics, anomaly_detection
|
| 123 |
+
- ✅ **Data (5 NEW)**: data_pipeline, statistical_analysis, time_series_forecasting, clustering_analysis, knowledge_graph
|
| 124 |
+
- ✅ **Code (10 NEW)**: code_translation, code_completion, code_explanation, code_minification, code_documentation_generation, code_duplication_detection, code_dead_code_analysis, code_complexity_analysis, code_dependency_analysis, bug_reproduction
|
| 125 |
+
- ✅ **System (4 NEW)**: system_design, api_design, graphql_skill, microservices
|
| 126 |
+
- ✅ **Language (5 NEW)**: prompt_engineering, sentiment_analysis, topic_modeling, language_detection, creative_writing
|
| 127 |
+
- ✅ **Cloud (1 NEW)**: cloud_deploy
|
| 128 |
+
- ✅ **Blockchain (1 NEW)**: blockchain_audit
|
| 129 |
+
- ✅ **Caching (1 NEW)**: caching_strategy
|
| 130 |
+
- ✅ **Classification (1 NEW)**: classification_automation
|
| 131 |
+
- ✅ **Regex (1 NEW)**: regex_master
|
| 132 |
+
- ✅ **Shell (1 NEW)**: shell_scripting
|
| 133 |
+
|
| 134 |
+
##### 🔧 Tools System (18+ → 80+)
|
| 135 |
+
- ✅ **Existing 24**: file_read/write/list/delete, shell_exec, python_exec, git_ops, http_request, web_fetch, web_search, code_search/lint/format, calculator, json/yaml/csv_parse, regex_search, archive, hash, encrypt, datetime, dns_lookup, ping
|
| 136 |
+
- ✅ **Database (12 NEW)**: sql_runner, sql_formatter, sql_migrator, postgres, mysql, sqlite, redis, mongo, elasticsearch, kafka, rabbitmq, graphql_client
|
| 137 |
+
- ✅ **DevOps/Cloud (12 NEW)**: docker, kubectl, terraform, ansible, aws_cli, gcloud_cli, azure_cli, ssh, scp, rsync, systemd, crontab
|
| 138 |
+
- ✅ **Code analysis (13 NEW)**: code_ast, code_complexity, code_dependency, code_metrics, code_smells, code_formatter_advanced, code_minifier, code_transpiler, code_runner, code_tester, code_compiler, code_profiler, code_coverage
|
| 139 |
+
- ✅ **Web/Network (12 NEW)**: websocket_client, grpc_client, url_shortener, dns_query, traceroute_tool, port_scanner, ssl_checker, ssl_generator, cert_checker, web_scraper, web_crawler, web_auth
|
| 140 |
+
- ✅ **Misc/Convert/Security (13 NEW)**: jwt_tool, oauth_tool, api_key_validator, markdown_converter, pdf_generator, image_processor, statistics_tool, linear_algebra_tool, probability_tool, ml_metrics_tool, model_evaluator, benchmark_runner, log_analyzer
|
| 141 |
+
|
| 142 |
+
##### 📊 Data Pipeline (5 → 8 sources, 60 → 500+ repos)
|
| 143 |
+
- ✅ `GitHubCollector` (expanded): 60+ → 500+ curated repos (Python, JS, TS, Go, Rust, C/C++, Java, C#, Ruby, PHP, Swift, Kotlin, ...)
|
| 144 |
+
- ✅ `HuggingFaceCollector` (expanded): 20+ → 150+ datasets (code, instruction, math, Vietnamese, multilingual)
|
| 145 |
+
- ✅ `ArxivCollector`: 20 → 40 queries
|
| 146 |
+
- ✅ `WikipediaCollector`: 18 → 50+ topics per language
|
| 147 |
+
- ✅ `StackOverflowCollector`: 30 → 47 tags
|
| 148 |
+
- ✅ `TheStackCollector` (v0.3 NEW): BigCode's The-Stack v2 (~600 languages)
|
| 149 |
+
- ✅ `StarCoder2Collector` (v0.3 NEW): github_code + commits + jupyter notebooks
|
| 150 |
+
- ✅ `PythonAlpacaCollector` (v0.3 NEW): aggregates 6 Python instruction datasets
|
| 151 |
+
|
| 152 |
+
##### 🧠 Processors (4 → 6)
|
| 153 |
+
- ✅ `TextCleaner`, `Deduplicator`, `QualityFilter`, `CodeFormatter` (existing)
|
| 154 |
+
- ✅ `LanguageIdProcessor` (v0.3 NEW): identifies vi/en/code, drops mislabeled
|
| 155 |
+
- ✅ `CodeQualityProcessor` (v0.3 NEW): scores Python 1-10 (docstring, type hints, no eval, etc.)
|
| 156 |
+
|
| 157 |
+
##### 🤝 Integrations (5 reference frameworks)
|
| 158 |
+
- ✅ `litgpt.py`: FusedLinear adapter (Apache 2.0, Lightning AI)
|
| 159 |
+
- ✅ `llamafactory.py`: dataset format converters (alpaca/sharegpt/chatml/completion → nexus)
|
| 160 |
+
- ✅ `axolotl.py`: AxolotlStyleConfig dataclass (typed training config schema)
|
| 161 |
+
- ✅ `openhands.py`: AgentLoop pattern (planner/executor/observer/reflector)
|
| 162 |
+
- ✅ `omp_gym.py`: OpenMP optimization benchmark tasks
|
| 163 |
+
|
| 164 |
+
##### 📈 Evaluation Module
|
| 165 |
+
- ✅ `BenchmarkSuite` - 10 benchmarks (HumanEval, MBPP, GSM8K, MMLU, BBH, MATH, ARC, TruthfulQA, AlpacaFarm, OMP-gym)
|
| 166 |
+
- ✅ Metrics: Perplexity, BLEU, ROUGE, F1, code-pass@k
|
| 167 |
+
|
| 168 |
+
#### 🔧 Cải tiến / Improvements
|
| 169 |
+
|
| 170 |
+
- ✅ **Auto-discovery registries**: Skills + Tools now scan directories dynamically — drop a `.py` file with a `Skill`/`Tool` subclass and it auto-registers
|
| 171 |
+
- ✅ **Stream-friendly training data**: `StreamingNexusDataset` for >1M example datasets (no RAM pressure)
|
| 172 |
+
- ✅ **Trimmed hardcoded data**: AUTHOR_TRAINING_DATA 150+ → 15 core examples (rest loaded from JSONL)
|
| 173 |
+
- ✅ **Lazy imports**: Faster startup; optional deps only imported when needed
|
| 174 |
+
- ✅ **Type hints**: Full typing throughout
|
| 175 |
+
- ✅ **Safety first**: All DANGEROUS/DESTRUCTIVE tools have `requires_confirmation=True` + `dry_run` support
|
| 176 |
+
- ✅ **Audit logging**: All tool calls logged to JSONL with timestamp, args, result, duration
|
| 177 |
+
- ✅ **Bilingual**: Vietnamese + English throughout
|
| 178 |
+
|
| 179 |
+
#### 📊 Thông số kỹ thuật / Technical Specs
|
| 180 |
+
|
| 181 |
+
| Thông số | v0.2 | v0.3 |
|
| 182 |
+
|----------|------|------|
|
| 183 |
+
| Version | 0.2.0 | 0.3.0 |
|
| 184 |
+
| Skills | 15 | 60+ |
|
| 185 |
+
| Tools | 18+ | 80+ |
|
| 186 |
+
| Data sources | 5 | 8 |
|
| 187 |
+
| Curated repos | 60+ | 500+ |
|
| 188 |
+
| Curated datasets | 20+ | 150+ |
|
| 189 |
+
| Configs | 5 | 7 |
|
| 190 |
+
| Reference frameworks | 0 | 5 |
|
| 191 |
+
| Attention backends | 1 (SDPA) | 3 (SDPA + FA2 + ALiBi) |
|
| 192 |
+
| Python version | 3.12.13 | 3.12.13 (strict) |
|
| 193 |
+
| PyTorch | >= 2.0 | >= 2.0 (>= 2.3 for 70b config) |
|
| 194 |
+
|
| 195 |
+
#### 📁 Cấu trúc thư mục v0.3 (key changes)
|
| 196 |
+
|
| 197 |
+
```
|
| 198 |
+
NexusCoder/
|
| 199 |
+
├── nexus/
|
| 200 |
+
│ ├── __init__.py # v0.3.0 metadata
|
| 201 |
+
│ ├── config.py # + 30b/70b configs + attention features
|
| 202 |
+
│ ├── model/
|
| 203 |
+
│ │ ├── attention.py # + FA2, ALiBi, SWA, QK-norm, KV quant
|
| 204 |
+
│ │ ├── rope.py # + NTK/YaRN scaling
|
| 205 |
+
│ │ ├── flash_attention.py # NEW
|
| 206 |
+
│ │ ├── alibi.py # NEW
|
| 207 |
+
│ │ ├── sliding_window.py # NEW
|
| 208 |
+
│ │ ├── layers.py # + MLP-parallel SwiGLU
|
| 209 |
+
│ │ └── transformer.py # + gradient checkpointing
|
| 210 |
+
│ ├── training/
|
| 211 |
+
│ │ └── dataset.py # trimmed + StreamingNexusDataset
|
| 212 |
+
│ ├── skills/ # 60+ skills, auto-discovery registry
|
| 213 |
+
│ ├── tools/ # 80+ tools, auto-discovery registry
|
| 214 |
+
│ ├── data/
|
| 215 |
+
│ │ ├── collectors/ # 8 collectors (3 NEW)
|
| 216 |
+
│ │ └── processors/ # 6 processors (2 NEW)
|
| 217 |
+
│ └── integrations/ # NEW: 5 reference framework adapters
|
| 218 |
+
├── configs/
|
| 219 |
+
│ ├── nexus_coder_30b.yaml # NEW
|
| 220 |
+
│ ├── nexus_coder_70b.yaml # NEW
|
| 221 |
+
│ └── sources.yaml # expanded to 500+ repos, 150+ datasets
|
| 222 |
+
├── ATTRIBUTIONS.md # NEW
|
| 223 |
+
├── requirements.txt # + 30 new optional deps
|
| 224 |
+
├── pyproject.toml # v0.3.0 + extras groups
|
| 225 |
+
└── setup.py # v0.3.0
|
| 226 |
+
```
|
| 227 |
+
|
| 228 |
+
#### 🚀 Migration từ v0.2
|
| 229 |
+
|
| 230 |
+
v0.3 backward compatible với v0.2:
|
| 231 |
+
- `NexusConfig()` vẫn hoạt động (default = large 10B)
|
| 232 |
+
- `NexusAgent()` vẫn hoạt động
|
| 233 |
+
- `AUTHOR_TRAINING_DATA` vẫn có (nhưng được tinh gọn)
|
| 234 |
+
- `scripts/train.py` vẫn hoạt động (nhưng có thêm config 30b, 70b)
|
| 235 |
+
|
| 236 |
+
Breaking changes (minor):
|
| 237 |
+
- `nexus.skills.registry._auto_register_defaults` giờ dùng dynamic discovery thay vì hardcoded imports
|
| 238 |
+
- `nexus.tools.registry._auto_register_defaults` tương tự
|
| 239 |
+
- `AUTHOR_TRAINING_DATA` giảm từ 150+ xuống 15 mẫu (phần còn lại load từ `data/processed/*.jsonl`)
|
| 240 |
+
|
| 241 |
+
#### 📦 Dependencies mới
|
| 242 |
+
|
| 243 |
+
```bash
|
| 244 |
+
# Database tools
|
| 245 |
+
pip install sqlalchemy psycopg2-binary pymysql redis pymongo elasticsearch kafka-python pika
|
| 246 |
+
|
| 247 |
+
# Web/Network tools
|
| 248 |
+
pip install aiohttp websockets grpcio beautifulsoup4 lxml
|
| 249 |
+
|
| 250 |
+
# DevOps tools
|
| 251 |
+
pip install paramiko kubernetes docker
|
| 252 |
+
|
| 253 |
+
# Media/Convert tools
|
| 254 |
+
pip install Pillow reportlab markdown
|
| 255 |
+
|
| 256 |
+
# ML tools
|
| 257 |
+
pip install scikit-learn scipy transformers accelerate peft
|
| 258 |
+
|
| 259 |
+
# Crypto
|
| 260 |
+
pip install pyjwt
|
| 261 |
+
|
| 262 |
+
# GPU acceleration
|
| 263 |
+
pip install flash-attn --no-build-isolation
|
| 264 |
+
|
| 265 |
+
# All at once
|
| 266 |
+
pip install -e ".[all]"
|
| 267 |
+
```
|
| 268 |
+
|
| 269 |
+
---
|
| 270 |
+
|
| 271 |
+
## v0.2.0 - 2026-08-16
|
| 272 |
+
|
| 273 |
+
### 🚀 Major Upgrade - Skills, Tools, và Data Pipeline
|
| 274 |
+
|
| 275 |
+
**Tác giả / Author**: Hieu Louis
|
| 276 |
+
|
| 277 |
+
#### ✨ Tính năng mới / New Features
|
| 278 |
+
|
| 279 |
+
##### 🎯 Skills System (15 skills)
|
| 280 |
+
- ✅ `code_generation` - Sinh code từ mô tả (Python, JS, Go, Rust, SQL, ...)
|
| 281 |
+
- ✅ `code_review` - Review code: bugs, security, performance
|
| 282 |
+
- ✅ `code_refactor` - Tái cấu trúc code (extract, rename, patterns)
|
| 283 |
+
- ✅ `debugging` - Debug đa ngôn ngữ với 7-step protocol
|
| 284 |
+
- ✅ `documentation` - Sinh docstrings, README, API docs
|
| 285 |
+
- ✅ `testing` - Unit/integration/E2E/property/mutation tests
|
| 286 |
+
- �� `algorithm_design` - Thiết kế thuật toán, complexity analysis
|
| 287 |
+
- ✅ `data_analysis` - EDA, statistics, visualization
|
| 288 |
+
- ✅ `translation` - Dịch song ngữ Việt-Anh
|
| 289 |
+
- ✅ `summarization` - Extractive + abstractive summarization
|
| 290 |
+
- ✅ `reasoning` - CoT, ToT, ReAct, self-consistency
|
| 291 |
+
- ✅ `math_skill` - Algebra, calculus, linear algebra, statistics
|
| 292 |
+
- ✅ `sql_generation` - SQL cho 7 dialects (Postgres, MySQL, ...)
|
| 293 |
+
- ✅ `security_audit` - OWASP Top 10, SAST, dependency scan
|
| 294 |
+
- ✅ `performance_optimization` - Profiling, bottleneck, optimization
|
| 295 |
+
|
| 296 |
+
##### 🔧 Tools System (15+ tools)
|
| 297 |
+
- ✅ `file_read` / `file_write` / `file_list` / `file_delete` - File operations
|
| 298 |
+
- ✅ `shell_exec` - Execute bash commands (sandboxed)
|
| 299 |
+
- ✅ `python_exec` - Execute Python code (restricted namespace)
|
| 300 |
+
- ✅ `git_ops` - Git commands với safety classification
|
| 301 |
+
- ✅ `http_request` - HTTP GET/POST/PUT/DELETE
|
| 302 |
+
- ✅ `web_fetch` - Fetch webpage, extract text
|
| 303 |
+
- ✅ `web_search` - Web search (Google/Bing/Brave API)
|
| 304 |
+
- ✅ `code_search` - Regex search trong code files
|
| 305 |
+
- ✅ `code_lint` / `code_format` - Lint & format code
|
| 306 |
+
- ✅ `calculator` - Safe math expression eval
|
| 307 |
+
- ✅ `json_parse` / `yaml_parse` / `csv_parse` - Data parsers
|
| 308 |
+
- ✅ `regex_search` - Regex search trong files
|
| 309 |
+
- ✅ `archive` - ZIP/TAR create/extract/list
|
| 310 |
+
- ✅ `hash` / `encrypt` - Hashing & AES-256-GCM encryption
|
| 311 |
+
- ✅ `datetime` - DateTime operations + timezone convert
|
| 312 |
+
- ✅ `dns_lookup` / `ping` - Network diagnostics
|
| 313 |
+
|
| 314 |
+
##### 📊 Training Data Pipeline
|
| 315 |
+
- ✅ `GitHubCollector` - Thu thập code từ 60+ curated GitHub repos
|
| 316 |
+
- ✅ `HuggingFaceCollector` - 20+ curated HF datasets (code, text, Vietnamese)
|
| 317 |
+
- ✅ `ArxivCollector` - Scientific papers từ arXiv API
|
| 318 |
+
- ✅ `WikipediaCollector` - Vietnamese + English Wikipedia
|
| 319 |
+
- ✅ `StackOverflowCollector` - Q&A từ StackOverflow API
|
| 320 |
+
- ✅ `TextCleaner` - HTML stripping, unicode normalize, whitespace cleanup
|
| 321 |
+
- ✅ `CodeFormatter` - Format code samples, detect language
|
| 322 |
+
- ✅ `Deduplicator` - MinHash LSH for near-duplicate detection
|
| 323 |
+
- ✅ `QualityFilter` - Quality scoring (length, diversity, repetition)
|
| 324 |
+
- ✅ `CurriculumLearning` - 4-stage curriculum (easy → expert)
|
| 325 |
+
|
| 326 |
+
---
|
| 327 |
+
|
| 328 |
+
## v0.1.0 - 2026-08-16
|
| 329 |
+
|
| 330 |
+
### 🎉 Initial Release - Foundation
|
| 331 |
+
|
| 332 |
+
**Tác giả / Author**: Hieu Louis
|
| 333 |
+
|
| 334 |
+
#### Thêm mới / Added
|
| 335 |
+
|
| 336 |
+
- ✅ Kiến trúc **Mixture of Experts (MoE)** với 24 experts, 3 active mỗi token
|
| 337 |
+
- ✅ Tổng **10 tỷ tham số (10B)** với chỉ **1.5 tỷ tham số active (1.5B)** mỗi token
|
| 338 |
+
- ✅ **Cửa sổ ngữ cảnh 50,000 tokens** với RoPE
|
| 339 |
+
- ✅ **Grouped Query Attention (GQA)** - 16 heads, 4 KV heads
|
| 340 |
+
- ✅ **RMSNorm** + **SwiGLU** activation
|
| 341 |
+
- ✅ **BPE Tokenizer** song ngữ Việt-Anh
|
| 342 |
+
- ✅ **Training script** với AdamW + cosine LR schedule
|
| 343 |
+
- ✅ **Inference engine** với top-k, top-p, temperature sampling
|
| 344 |
+
- ✅ **AI Agent wrapper** (Nexus Agent) với quản lý hội thoại
|
| 345 |
+
- ✅ **Hardcoded author info** - model luôn nhớ được tạo bởi Hieu Louis
|
| 346 |
+
- ✅ **Test suite** đầy đủ
|
| 347 |
+
- ✅ **Song ngữ Việt-Anh** trong README và giao tiếp
|
| 348 |
+
- ✅ **MIT License**
|
| 349 |
+
- ✅ Tương thích **Python 3.12.13**
|
CONTRIBUTING.md
ADDED
|
@@ -0,0 +1,66 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Contributing to Nexus Coder
|
| 2 |
+
|
| 3 |
+
Thanks for your interest in contributing! This project is an open AI
|
| 4 |
+
architecture in active development. Both humans and AI agents are welcome.
|
| 5 |
+
|
| 6 |
+
> **AI agents:** read `AGENTS.md` first — it is written specifically for you.
|
| 7 |
+
|
| 8 |
+
## Code of Conduct
|
| 9 |
+
|
| 10 |
+
Be respectful. This project is built by a small team with limited resources.
|
| 11 |
+
Good-faith contributions are valued; trolling, spamming, or fake claims are not.
|
| 12 |
+
|
| 13 |
+
## What We Need Help With
|
| 14 |
+
|
| 15 |
+
1. **Running the small configs** — verify `tiny` / `small` train and run on CPU.
|
| 16 |
+
2. **Testing skills & tools** — exercise `nexus/skills/` and `nexus/tools/`.
|
| 17 |
+
3. **Reviewing integrations** — verify patterns adapted from upstream projects.
|
| 18 |
+
4. **Tests** — `tests/` is thin; add coverage for model layers, tokenizer, tools.
|
| 19 |
+
5. **Docs** — architecture docs always need improvement.
|
| 20 |
+
6. **Training experiments** — if you have GPUs, try a small real training run
|
| 21 |
+
and report honestly what you observed.
|
| 22 |
+
|
| 23 |
+
## Getting Started
|
| 24 |
+
|
| 25 |
+
```bash
|
| 26 |
+
git clone https://github.com/mhieuhonda/NexusCoder.git
|
| 27 |
+
cd NexusCoder
|
| 28 |
+
python3.12.13 -m venv venv
|
| 29 |
+
source venv/bin/activate
|
| 30 |
+
pip install -r requirements.txt
|
| 31 |
+
```
|
| 32 |
+
|
| 33 |
+
Python version is **3.12.13 (strict)**. Use `pyenv` or similar to match it.
|
| 34 |
+
|
| 35 |
+
## Contribution Workflow
|
| 36 |
+
|
| 37 |
+
1. **Open an issue first** describing what you plan to do (check for existing
|
| 38 |
+
ones to avoid duplication).
|
| 39 |
+
2. **Fork the repo** and create a branch.
|
| 40 |
+
3. Make your changes, keeping them **small and focused**.
|
| 41 |
+
4. **Verify** your change locally before opening a PR.
|
| 42 |
+
5. Open the **pull request** and describe what you did and how you verified it.
|
| 43 |
+
|
| 44 |
+
## Style
|
| 45 |
+
|
| 46 |
+
- Follow the existing code style in the file you are touching.
|
| 47 |
+
- Add or update tests for any new code.
|
| 48 |
+
- Keep commit messages clear and descriptive.
|
| 49 |
+
|
| 50 |
+
## Labels
|
| 51 |
+
|
| 52 |
+
- `good first issue` — beginner-friendly tasks (agents: start here)
|
| 53 |
+
- `help wanted` — tasks where maintainers explicitly want outside help
|
| 54 |
+
- `bug` — something is broken
|
| 55 |
+
- `enhancement` — new feature or improvement
|
| 56 |
+
|
| 57 |
+
## License & Attribution
|
| 58 |
+
|
| 59 |
+
Contributions are licensed under **NAL-1.0** (Attribution Required). By
|
| 60 |
+
contributing, you agree your changes are covered by this license and that the
|
| 61 |
+
original author **Hieu Louis** (github.com/mhieuhonda) retains attribution
|
| 62 |
+
requirements. See `LICENSE` and `ATTRIBUTIONS.md`.
|
| 63 |
+
|
| 64 |
+
## Questions
|
| 65 |
+
|
| 66 |
+
Open an issue, or reach out through the **code-realm** community on Moltbook.
|
LICENSE
ADDED
|
@@ -0,0 +1,193 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
NexusCoder Attribution License v1.0 (NAL-1.0)
|
| 2 |
+
==============================================
|
| 3 |
+
Copyright (c) 2026 Hieu Louis (https://github.com/mhieuhonda)
|
| 4 |
+
|
| 5 |
+
This license applies to the Nexus Coder project, including all source code,
|
| 6 |
+
configuration files, documentation, model architecture, training methodology,
|
| 7 |
+
and associated materials contained in this repository.
|
| 8 |
+
|
| 9 |
+
By exercising any rights granted by this license, you accept and agree to be
|
| 10 |
+
bound by its terms and conditions.
|
| 11 |
+
|
| 12 |
+
----------------------------------------------------------------------
|
| 13 |
+
|
| 14 |
+
1. DEFINITIONS
|
| 15 |
+
|
| 16 |
+
"Project" means the Nexus Coder project, including all software, model
|
| 17 |
+
architecture code, training scripts, configurations, documentation, and
|
| 18 |
+
data pipeline code contained in this repository.
|
| 19 |
+
|
| 20 |
+
"Author" means Hieu Louis, the original creator of the Project
|
| 21 |
+
(GitHub: https://github.com/mhieuhonda).
|
| 22 |
+
|
| 23 |
+
"Derivative Work" means any work, model, software, or artifact that is
|
| 24 |
+
based on, derived from, or incorporates any part of the Project, including
|
| 25 |
+
but not limited to:
|
| 26 |
+
- Fine-tuned or modified versions of the Project
|
| 27 |
+
- Models trained using the Project's architecture or methodology
|
| 28 |
+
- Software that redistributes, modifies, or builds upon the Project
|
| 29 |
+
- Repackaged versions of the Project, in whole or in part
|
| 30 |
+
|
| 31 |
+
"Attribution" means clearly and prominently crediting the Author as
|
| 32 |
+
the original creator of the Project, in the manner specified in
|
| 33 |
+
Section 3 below.
|
| 34 |
+
|
| 35 |
+
"You" or "Your" means any person or entity exercising rights under
|
| 36 |
+
this license.
|
| 37 |
+
|
| 38 |
+
----------------------------------------------------------------------
|
| 39 |
+
|
| 40 |
+
2. GRANTED RIGHTS
|
| 41 |
+
|
| 42 |
+
Subject to the terms of this license, the Author grants You a worldwide,
|
| 43 |
+
royalty-free, non-exclusive, perpetual license to:
|
| 44 |
+
|
| 45 |
+
(a) Use, copy, modify, merge, publish, distribute, sublicense, and/or
|
| 46 |
+
sell copies of the Project, in whole or in part.
|
| 47 |
+
|
| 48 |
+
(b) Train, fine-tune, distill, prune, quantize, or otherwise create
|
| 49 |
+
Derivative Works based on the Project, for any commercial or
|
| 50 |
+
non-commercial purpose.
|
| 51 |
+
|
| 52 |
+
(c) Use the Project's architecture, methodology, training pipeline,
|
| 53 |
+
code corpus, or any other component to build Your own products,
|
| 54 |
+
services, research, or any other work.
|
| 55 |
+
|
| 56 |
+
(d) Distribute Derivative Works under any license You choose, provided
|
| 57 |
+
that You comply with the Attribution requirement (Section 3).
|
| 58 |
+
|
| 59 |
+
----------------------------------------------------------------------
|
| 60 |
+
|
| 61 |
+
3. ATTRIBUTION REQUIREMENT (MANDATORY)
|
| 62 |
+
|
| 63 |
+
You MUST attribute the Author (Hieu Louis) as the original creator of
|
| 64 |
+
the Project in all of the following circumstances:
|
| 65 |
+
|
| 66 |
+
(a) REDISTRIBUTION: When You distribute, publish, or make available
|
| 67 |
+
the Project (or any Derivative Work), You must include:
|
| 68 |
+
- The Author's name: "Hieu Louis"
|
| 69 |
+
- A link to the original project:
|
| 70 |
+
https://github.com/mhieuhonda/NexusCoder
|
| 71 |
+
- A notice that the work is based on or derived from the Project
|
| 72 |
+
|
| 73 |
+
(b) MODELS TRAINED USING THE PROJECT: If You train, fine-tune, or
|
| 74 |
+
otherwise create a model using the Project's architecture,
|
| 75 |
+
methodology, training pipeline, code, or any other component:
|
| 76 |
+
- You MUST include in the model card, README, documentation,
|
| 77 |
+
or any other accompanying material:
|
| 78 |
+
"Built using Nexus Coder by Hieu Louis
|
| 79 |
+
(https://github.com/mhieuhonda/NexusCoder)"
|
| 80 |
+
- This attribution MUST be visible to end users of the model,
|
| 81 |
+
including in API responses, UI, model cards, or download pages
|
| 82 |
+
where reasonable and customary.
|
| 83 |
+
|
| 84 |
+
(c) PRODUCTS & SERVICES: If You build a product, service, or application
|
| 85 |
+
that uses the Project or any Derivative Work:
|
| 86 |
+
- You MUST include in the product's documentation, About page,
|
| 87 |
+
or credits section: "Powered by Nexus Coder by Hieu Louis"
|
| 88 |
+
- If the product has an "About" or "Credits" UI element,
|
| 89 |
+
the attribution must appear there.
|
| 90 |
+
|
| 91 |
+
(d) RESEARCH PUBLICATIONS: If You publish research that used the
|
| 92 |
+
Project, You MUST cite:
|
| 93 |
+
Hieu Louis. "Nexus Coder: AI Code & Security Engine (CyberForge
|
| 94 |
+
Edition)." https://github.com/mhieuhonda/NexusCoder, 2026.
|
| 95 |
+
|
| 96 |
+
(e) FORKED REPOSITORIES: If You fork the Project on GitHub or any
|
| 97 |
+
similar platform:
|
| 98 |
+
- You MUST keep the attribution in the README and LICENSE
|
| 99 |
+
- You MUST NOT claim to be the original author
|
| 100 |
+
- You MAY add Your own authorship for Your own contributions
|
| 101 |
+
|
| 102 |
+
----------------------------------------------------------------------
|
| 103 |
+
|
| 104 |
+
4. ATTRIBUTION FORMAT
|
| 105 |
+
|
| 106 |
+
The attribution must be clear, visible, and accessible to end users.
|
| 107 |
+
Acceptable formats include (but are not limited to):
|
| 108 |
+
|
| 109 |
+
Short form (for UI, API responses, footers):
|
| 110 |
+
"Powered by Nexus Coder by Hieu Louis"
|
| 111 |
+
|
| 112 |
+
Medium form (for README, docs):
|
| 113 |
+
"Built using Nexus Coder by Hieu Louis
|
| 114 |
+
(https://github.com/mhieuhonda/NexusCoder)"
|
| 115 |
+
|
| 116 |
+
Full form (for model cards, academic publications):
|
| 117 |
+
"This work is based on Nexus Coder (v0.4.0, CyberForge Edition),
|
| 118 |
+
created by Hieu Louis (https://github.com/mhieuhonda/NexusCoder)
|
| 119 |
+
and licensed under NAL-1.0."
|
| 120 |
+
|
| 121 |
+
----------------------------------------------------------------------
|
| 122 |
+
|
| 123 |
+
5. NO WARRANTIES
|
| 124 |
+
|
| 125 |
+
The Project is provided "AS IS", without warranty of any kind, express
|
| 126 |
+
or implied, including but not limited to the warranties of
|
| 127 |
+
merchantability, fitness for a particular purpose, and non-infringement.
|
| 128 |
+
In no event shall the Author be liable for any claim, damages, or
|
| 129 |
+
other liability, whether in an action of contract, tort, or otherwise,
|
| 130 |
+
arising from, out of, or in connection with the Project or the use or
|
| 131 |
+
other dealings in the Project.
|
| 132 |
+
|
| 133 |
+
----------------------------------------------------------------------
|
| 134 |
+
|
| 135 |
+
6. NO ENDORSEMENT
|
| 136 |
+
|
| 137 |
+
You MUST NOT use the Author's name, the Project's name, or any
|
| 138 |
+
associated trademarks to imply endorsement of Your product, service,
|
| 139 |
+
or research without prior written permission from the Author.
|
| 140 |
+
|
| 141 |
+
----------------------------------------------------------------------
|
| 142 |
+
|
| 143 |
+
7. NON-INTERFERENCE WITH ATTRIBUTION
|
| 144 |
+
|
| 145 |
+
You MUST NOT remove, obscure, or alter any attribution notices
|
| 146 |
+
included in the Project. You MUST NOT implement technical measures
|
| 147 |
+
(e.g., watermark removal, fine-tuning that erases embedded authorship
|
| 148 |
+
information) that would have the effect of obscuring or removing the
|
| 149 |
+
Author's attribution.
|
| 150 |
+
|
| 151 |
+
----------------------------------------------------------------------
|
| 152 |
+
|
| 153 |
+
8. TERMINATION
|
| 154 |
+
|
| 155 |
+
Your rights under this license terminate automatically if You fail to
|
| 156 |
+
comply with any of its terms, especially the Attribution requirement
|
| 157 |
+
(Section 3). Upon termination, You must cease all use and distribution
|
| 158 |
+
of the Project and any Derivative Works, and destroy all copies in
|
| 159 |
+
Your possession or control.
|
| 160 |
+
|
| 161 |
+
----------------------------------------------------------------------
|
| 162 |
+
|
| 163 |
+
9. VERSIONING
|
| 164 |
+
|
| 165 |
+
This is version 1.0 of the NexusCoder Attribution License ("NAL-1.0").
|
| 166 |
+
Future versions of the license, if any, will be designated by incrementing
|
| 167 |
+
the version number. The Author may release updated versions of this
|
| 168 |
+
license to address new use cases or clarify existing terms, but such
|
| 169 |
+
updates will not retroactively change the terms under which You received
|
| 170 |
+
the Project unless You explicitly choose to adopt the new version.
|
| 171 |
+
|
| 172 |
+
----------------------------------------------------------------------
|
| 173 |
+
|
| 174 |
+
10. ENTIRE AGREEMENT
|
| 175 |
+
|
| 176 |
+
This license constitutes the entire agreement between You and the
|
| 177 |
+
Author with respect to the Project. If any provision of this license
|
| 178 |
+
is held to be unenforceable, the remaining provisions shall remain
|
| 179 |
+
in full force and effect.
|
| 180 |
+
|
| 181 |
+
----------------------------------------------------------------------
|
| 182 |
+
|
| 183 |
+
For questions or to request alternative licensing terms, contact:
|
| 184 |
+
|
| 185 |
+
Hieu Louis
|
| 186 |
+
GitHub: https://github.com/mhieuhonda
|
| 187 |
+
Year: 2026
|
| 188 |
+
|
| 189 |
+
----------------------------------------------------------------------
|
| 190 |
+
|
| 191 |
+
By using, copying, modifying, distributing, or training on the Project,
|
| 192 |
+
You acknowledge that You have read, understood, and agree to be bound by
|
| 193 |
+
the terms of this NexusCoder Attribution License v1.0.
|
README.md
ADDED
|
@@ -0,0 +1,134 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
<div align="center">
|
| 2 |
+
|
| 3 |
+
# 🧠 Nexus Coder
|
| 4 |
+
|
| 5 |
+
### AI Code & Security Engine — CyberForge Edition
|
| 6 |
+
|
| 7 |
+
**An open architecture for next‑generation code generation and security analysis**
|
| 8 |
+
|
| 9 |
+
[](https://www.python.org/)
|
| 10 |
+
[](https://pytorch.org/)
|
| 11 |
+
[](LICENSE)
|
| 12 |
+
[]()
|
| 13 |
+
[]()
|
| 14 |
+
[](https://github.com/mhieuhonda/NexusCoder)
|
| 15 |
+
[](https://github.com/mhieuhonda/NexusCoder)
|
| 16 |
+
[](https://github.com/mhieuhonda/NexusCoder)
|
| 17 |
+
|
| 18 |
+
**Created by [Hieu Louis](https://github.com/mhieuhonda)** · 2026
|
| 19 |
+
|
| 20 |
+
</div>
|
| 21 |
+
|
| 22 |
+
## 📖 Introduction
|
| 23 |
+
|
| 24 |
+
**Nexus Coder** is an open‑source AI architecture, designed from the ground up by **Hieu Louis**, focused on two core capabilities:
|
| 25 |
+
|
| 26 |
+
- **High‑quality code generation** powered by a large‑scale Mixture‑of‑Experts (MoE) Transformer.
|
| 27 |
+
- **Deep security analysis** for source code and systems.
|
| 28 |
+
|
| 29 |
+
The project is under **active development**. This repository provides:
|
| 30 |
+
|
| 31 |
+
- The complete **model architecture source code** (Python/PyTorch).
|
| 32 |
+
- A **data collection and processing pipeline** for code from multiple sources.
|
| 33 |
+
- A **multi‑stage training framework** designed to scale.
|
| 34 |
+
- **60+ skills** and **80+ tools** with automatic registration.
|
| 35 |
+
- Configurations ranging from `tiny` (5M) to `423b` (423B parameters).
|
| 36 |
+
|
| 37 |
+
> **Important:** The model is **not pretrained** yet. We distribute only the architecture source and training pipeline. Users need to train their own models on their own data, in compliance with the NAL‑1.0 license.
|
| 38 |
+
|
| 39 |
+
## 📊 Key Technical Specifications
|
| 40 |
+
|
| 41 |
+
| Item | Value |
|
| 42 |
+
|------|-------|
|
| 43 |
+
| Total parameters | ~423B |
|
| 44 |
+
| Active parameters per token | ~39B |
|
| 45 |
+
| Context window | 3,000,000 tokens (3M) |
|
| 46 |
+
| Architecture | MoE Transformer (GQA + RoPE/YaRN + RMSNorm + SwiGLU + FlashAttention‑2 + Sliding Window + QK‑norm + KV cache quantization + MLP‑parallel + Gradient checkpointing) |
|
| 47 |
+
| Skills | 60+ (code, devops, ML, data, security, cloud, system, blockchain, language) |
|
| 48 |
+
| Tools | 80+ (file, exec, web, code analysis, database, devops, crypto, math, network) |
|
| 49 |
+
| Data sources | 8+ (GitHub curated corpus, HuggingFace, arXiv, Wikipedia, StackOverflow, The‑Stack v2, StarCoder2‑data, Python‑Alpaca) |
|
| 50 |
+
| Python version | 3.12.13 (strict) |
|
| 51 |
+
|
| 52 |
+
## 🚀 Quick Install
|
| 53 |
+
|
| 54 |
+
```bash
|
| 55 |
+
git clone https://github.com/mhieuhonda/NexusCoder.git
|
| 56 |
+
cd NexusCoder
|
| 57 |
+
python3.12.13 -m venv venv
|
| 58 |
+
source venv/bin/activate
|
| 59 |
+
pip install -r requirements.txt
|
| 60 |
+
# or: pip install -e ".[all]"
|
| 61 |
+
```
|
| 62 |
+
|
| 63 |
+
💻 Usage
|
| 64 |
+
|
| 65 |
+
```bash
|
| 66 |
+
# Print configuration summary
|
| 67 |
+
python -c "from nexus.config import print_config_summary; print_config_summary()"
|
| 68 |
+
|
| 69 |
+
# Tiny demo (CPU)
|
| 70 |
+
python scripts/train.py --config tiny --steps 100
|
| 71 |
+
|
| 72 |
+
# Train larger configurations (requires GPU)
|
| 73 |
+
python scripts/train.py --config large --steps 5000 --use-amp
|
| 74 |
+
python scripts/train.py --config 423b --steps 50000 --use-amp --deepspeed
|
| 75 |
+
```
|
| 76 |
+
|
| 77 |
+
📁 Project Structure
|
| 78 |
+
|
| 79 |
+
```
|
| 80 |
+
NexusCoder/
|
| 81 |
+
├── nexus/ # Main package
|
| 82 |
+
│ ├── model/ # MoE Transformer (attention, MoE, layers, ...)
|
| 83 |
+
│ ├── tokenizer/
|
| 84 |
+
│ ├── training/ # Trainer + Dataset
|
| 85 |
+
│ ├── inference/
|
| 86 |
+
│ ├── agent/ # Planner, Router, Memory, Safety
|
| 87 |
+
│ ├── skills/ # 60+ skills (auto‑discovery)
|
| 88 |
+
│ ├── tools/ # 80+ tools (auto‑discovery)
|
| 89 |
+
│ ├── data/ # Collectors + Processors
|
| 90 |
+
│ ├── optim/ # Quantize, LoRA, Distill, Prune
|
| 91 |
+
│ ├── safety/ # Filters, Guardrails
|
| 92 |
+
│ ├── eval/ # Benchmarks, Metrics
|
| 93 |
+
│ ├── integrations/ # litgpt, LlamaFactory, axolotl, OpenHands, omp‑gym
|
| 94 |
+
│ └── utils/
|
| 95 |
+
├── configs/ # YAML configs (tiny → 423B)
|
| 96 |
+
├── scripts/ # CLI scripts
|
| 97 |
+
├── docs/ # ARCHITECTURE, TRAINING, SKILLS, TOOLS, DATA
|
| 98 |
+
├── tests/
|
| 99 |
+
├── ATTRIBUTIONS.md
|
| 100 |
+
├── CHANGELOG.md
|
| 101 |
+
├── LICENSE # NAL‑1.0 (Attribution Required)
|
| 102 |
+
├── requirements.txt
|
| 103 |
+
├── pyproject.toml
|
| 104 |
+
├── setup.py
|
| 105 |
+
└── README.md
|
| 106 |
+
```
|
| 107 |
+
|
| 108 |
+
⚖️ License
|
| 109 |
+
|
| 110 |
+
Released under the NexusCoder Attribution License v1.0 (NAL‑1.0).
|
| 111 |
+
|
| 112 |
+
· You may use, modify, distribute, and train models for any purpose.
|
| 113 |
+
· Attribution is required to the original author: Hieu Louis (github.com/mhieuhonda).
|
| 114 |
+
· No warranty. See LICENSE for details.
|
| 115 |
+
|
| 116 |
+
👤 Author
|
| 117 |
+
|
| 118 |
+
<div align="center">
|
| 119 |
+
|
| 120 |
+
Hieu Louis · 2026
|
| 121 |
+
|
| 122 |
+
· GitHub: @mhieuhonda
|
| 123 |
+
· Project: NexusCoder
|
| 124 |
+
· License: NAL‑1.0 (Attribution Required)
|
| 125 |
+
|
| 126 |
+
</div>
|
| 127 |
+
|
| 128 |
+
<div align="center">
|
| 129 |
+
|
| 130 |
+
Nexus Coder — CyberForge Edition
|
| 131 |
+
|
| 132 |
+
Made by Hieu Louis · 2026
|
| 133 |
+
|
| 134 |
+
</div>
|
configs/code_corpus.yaml
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
configs/nexus_coder_10b.yaml
ADDED
|
@@ -0,0 +1,120 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Nexus Coder Configuration - Large (10B/1.5B) v0.3 - DEFAULT
|
| 2 |
+
# Author: Hieu Louis (2026)
|
| 3 |
+
# Default model. 32+ GPU recommended for full pretrain.
|
| 4 |
+
|
| 5 |
+
model:
|
| 6 |
+
name: "Nexus Coder"
|
| 7 |
+
agent_name: "Nexus"
|
| 8 |
+
author: "Hieu Louis"
|
| 9 |
+
version: "0.3.0"
|
| 10 |
+
github: "mhieuhonda"
|
| 11 |
+
year: "2026"
|
| 12 |
+
|
| 13 |
+
architecture:
|
| 14 |
+
vocab_size: 32000
|
| 15 |
+
hidden_size: 2048
|
| 16 |
+
num_hidden_layers: 12
|
| 17 |
+
num_attention_heads: 16
|
| 18 |
+
num_kv_heads: 4 # Grouped Query Attention
|
| 19 |
+
head_dim: 128
|
| 20 |
+
intermediate_size: 5632 # per-expert
|
| 21 |
+
hidden_act: "silu" # SwiGLU
|
| 22 |
+
norm_type: "rmsnorm"
|
| 23 |
+
|
| 24 |
+
moe:
|
| 25 |
+
num_experts: 24 # Tổng số chuyên gia
|
| 26 |
+
num_active_experts: 3 # Chuyên gia kích hoạt mỗi token
|
| 27 |
+
router_aux_loss_coef: 0.001
|
| 28 |
+
router_jitter_noise: 0.0
|
| 29 |
+
|
| 30 |
+
context:
|
| 31 |
+
max_position_embeddings: 50000 # 50k tokens
|
| 32 |
+
rotary_emb_base: 10000.0
|
| 33 |
+
rope_scaling_type: null
|
| 34 |
+
rope_scaling_factor: 1.0
|
| 35 |
+
|
| 36 |
+
# v0.3 NEW attention features
|
| 37 |
+
attention:
|
| 38 |
+
use_flash_attention: true # PyTorch SDPA
|
| 39 |
+
use_flash_attention_2: false # FlashAttention-2 (optional, install flash-attn)
|
| 40 |
+
use_alibi: false # ALiBi alternative to RoPE
|
| 41 |
+
alibi_max_slope: 8.0
|
| 42 |
+
use_sliding_window: true # alternating SWA / global layers
|
| 43 |
+
sliding_window_size: 4096
|
| 44 |
+
sliding_window_layers: null # null = alternate even/odd layers
|
| 45 |
+
use_qk_norm: true # RMSNorm on Q and K (Llama-3 style)
|
| 46 |
+
qk_norm_eps: 1.0e-6
|
| 47 |
+
mlp_parallel: true # fused gate+up projection
|
| 48 |
+
|
| 49 |
+
compute:
|
| 50 |
+
use_kv_cache: true
|
| 51 |
+
kv_cache_quantization: null # null | "int8" | "fp8"
|
| 52 |
+
gradient_checkpointing: false
|
| 53 |
+
tensor_parallel_size: 1
|
| 54 |
+
pipeline_parallel_size: 1
|
| 55 |
+
expert_parallel_size: 1
|
| 56 |
+
sequence_parallel: false
|
| 57 |
+
|
| 58 |
+
params:
|
| 59 |
+
total: "~10.22B"
|
| 60 |
+
active: "~1.50B"
|
| 61 |
+
expert_utilization: "12.5%"
|
| 62 |
+
estimated_disk_mb_fp16: 19500
|
| 63 |
+
estimated_disk_mb_int8: 9750
|
| 64 |
+
estimated_disk_mb_int4: 4875
|
| 65 |
+
|
| 66 |
+
training:
|
| 67 |
+
learning_rate: 5.0e-4
|
| 68 |
+
weight_decay: 0.01
|
| 69 |
+
warmup_steps: 100
|
| 70 |
+
max_steps: 5000
|
| 71 |
+
per_device_batch_size: 4
|
| 72 |
+
gradient_accumulation_steps: 4
|
| 73 |
+
logging_steps: 10
|
| 74 |
+
save_steps: 500
|
| 75 |
+
max_grad_norm: 1.0
|
| 76 |
+
seed: 42
|
| 77 |
+
use_amp: true
|
| 78 |
+
|
| 79 |
+
inference:
|
| 80 |
+
max_new_tokens: 200
|
| 81 |
+
temperature: 0.8
|
| 82 |
+
top_k: 50
|
| 83 |
+
top_p: 0.9
|
| 84 |
+
do_sample: true
|
| 85 |
+
|
| 86 |
+
personality:
|
| 87 |
+
type: "humorous"
|
| 88 |
+
language: "bilingual"
|
| 89 |
+
specialties:
|
| 90 |
+
- programming
|
| 91 |
+
- conversation
|
| 92 |
+
- devops
|
| 93 |
+
- ml
|
| 94 |
+
- security
|
| 95 |
+
|
| 96 |
+
# v0.3 NEW capabilities
|
| 97 |
+
capabilities:
|
| 98 |
+
skills_count: 60
|
| 99 |
+
tools_count: 80
|
| 100 |
+
data_sources:
|
| 101 |
+
- github
|
| 102 |
+
- huggingface
|
| 103 |
+
- arxiv
|
| 104 |
+
- wikipedia
|
| 105 |
+
- stackoverflow
|
| 106 |
+
- the_stack
|
| 107 |
+
- starcoder2_data
|
| 108 |
+
- python_alpaca
|
| 109 |
+
training_frameworks_referenced:
|
| 110 |
+
- litgpt
|
| 111 |
+
- llamafactory
|
| 112 |
+
- axolotl
|
| 113 |
+
- openhands
|
| 114 |
+
- omp_gym
|
| 115 |
+
|
| 116 |
+
environment:
|
| 117 |
+
python_version: "3.12.13"
|
| 118 |
+
pytorch_version: ">=2.0"
|
| 119 |
+
cuda_required: false # có thể chạy trên CPU (chậm)
|
| 120 |
+
recommended_gpus: "32+ H100 80GB for full pretrain"
|
configs/nexus_coder_30b.yaml
ADDED
|
@@ -0,0 +1,103 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Nexus Coder Configuration - 30B/3B v0.3 NEW
|
| 2 |
+
# Pretrain on 64-128 H100 80GB GPUs.
|
| 3 |
+
# Recommended for serious pretraining at frontier scale.
|
| 4 |
+
# Author: Hieu Louis (2026)
|
| 5 |
+
|
| 6 |
+
model:
|
| 7 |
+
name: "Nexus Coder 30B"
|
| 8 |
+
agent_name: "Nexus"
|
| 9 |
+
author: "Hieu Louis"
|
| 10 |
+
version: "0.3.0-30b"
|
| 11 |
+
github: "mhieuhonda"
|
| 12 |
+
year: "2026"
|
| 13 |
+
|
| 14 |
+
architecture:
|
| 15 |
+
vocab_size: 64000
|
| 16 |
+
hidden_size: 4096
|
| 17 |
+
num_hidden_layers: 24
|
| 18 |
+
num_attention_heads: 32
|
| 19 |
+
num_kv_heads: 8
|
| 20 |
+
head_dim: 128
|
| 21 |
+
intermediate_size: 11264
|
| 22 |
+
hidden_act: "silu"
|
| 23 |
+
norm_type: "rmsnorm"
|
| 24 |
+
|
| 25 |
+
moe:
|
| 26 |
+
num_experts: 48
|
| 27 |
+
num_active_experts: 4
|
| 28 |
+
router_aux_loss_coef: 0.001
|
| 29 |
+
router_jitter_noise: 0.0
|
| 30 |
+
|
| 31 |
+
context:
|
| 32 |
+
max_position_embeddings: 65536
|
| 33 |
+
rotary_emb_base: 10000.0
|
| 34 |
+
rope_scaling_type: "dynamic" # NTK-aware scaling for 2× context
|
| 35 |
+
rope_scaling_factor: 2.0
|
| 36 |
+
|
| 37 |
+
attention:
|
| 38 |
+
use_flash_attention: true
|
| 39 |
+
use_flash_attention_2: true # mandatory at this scale
|
| 40 |
+
use_alibi: false
|
| 41 |
+
use_sliding_window: true
|
| 42 |
+
sliding_window_size: 8192
|
| 43 |
+
use_qk_norm: true
|
| 44 |
+
qk_norm_eps: 1.0e-6
|
| 45 |
+
mlp_parallel: true
|
| 46 |
+
|
| 47 |
+
compute:
|
| 48 |
+
use_kv_cache: true
|
| 49 |
+
kv_cache_quantization: "int8"
|
| 50 |
+
gradient_checkpointing: true
|
| 51 |
+
tensor_parallel_size: 4
|
| 52 |
+
pipeline_parallel_size: 1
|
| 53 |
+
expert_parallel_size: 4
|
| 54 |
+
sequence_parallel: false
|
| 55 |
+
|
| 56 |
+
params:
|
| 57 |
+
total: "~30B"
|
| 58 |
+
active: "~3B"
|
| 59 |
+
expert_utilization: "8.3%"
|
| 60 |
+
estimated_disk_mb_fp16: 60000
|
| 61 |
+
estimated_disk_mb_int8: 30000
|
| 62 |
+
estimated_disk_mb_int4: 15000
|
| 63 |
+
kv_cache_mb_per_token_fp16: 0.019
|
| 64 |
+
kv_cache_mb_per_token_int8: 0.0095
|
| 65 |
+
|
| 66 |
+
training:
|
| 67 |
+
learning_rate: 2.0e-4
|
| 68 |
+
weight_decay: 0.01
|
| 69 |
+
warmup_steps: 500
|
| 70 |
+
max_steps: 10000
|
| 71 |
+
per_device_batch_size: 1
|
| 72 |
+
gradient_accumulation_steps: 32
|
| 73 |
+
logging_steps: 10
|
| 74 |
+
save_steps: 1000
|
| 75 |
+
max_grad_norm: 1.0
|
| 76 |
+
seed: 42
|
| 77 |
+
use_amp: true
|
| 78 |
+
use_deepspeed: true
|
| 79 |
+
deepspeed_config: "configs/ds_config_zero3.json"
|
| 80 |
+
total_tokens_target: 500_000_000_000 # 500B tokens
|
| 81 |
+
|
| 82 |
+
inference:
|
| 83 |
+
max_new_tokens: 1000
|
| 84 |
+
temperature: 0.7
|
| 85 |
+
top_k: 50
|
| 86 |
+
top_p: 0.9
|
| 87 |
+
do_sample: true
|
| 88 |
+
|
| 89 |
+
personality:
|
| 90 |
+
type: "humorous"
|
| 91 |
+
language: "bilingual"
|
| 92 |
+
|
| 93 |
+
capabilities:
|
| 94 |
+
skills_count: 60
|
| 95 |
+
tools_count: 80
|
| 96 |
+
|
| 97 |
+
environment:
|
| 98 |
+
python_version: "3.12.13"
|
| 99 |
+
pytorch_version: ">=2.0"
|
| 100 |
+
cuda_required: true
|
| 101 |
+
min_gpu_memory_gb: 80
|
| 102 |
+
recommended_gpus: "64-128 H100 80GB"
|
| 103 |
+
estimated_training_time: "~30 days on 64 H100s"
|
configs/nexus_coder_423b.yaml
ADDED
|
@@ -0,0 +1,89 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# ============================================================================
|
| 2 |
+
# Nexus Coder v0.4 — CyberForge Config (423B / 39B / 3M context)
|
| 3 |
+
# ============================================================================
|
| 4 |
+
# Supreme variant — CyberGym training hooks enabled by default.
|
| 5 |
+
# Math (verified):
|
| 6 |
+
# embed (200k × 7168) = 1.43B
|
| 7 |
+
# per_layer_total (48 exp) = 17.03B
|
| 8 |
+
# per_layer_active (4 exp) = 1.53B
|
| 9 |
+
# 24 layers = 408B total / 36.6B active
|
| 10 |
+
# + LM head + norms + routers = ~412-423B total / ~39.5B active
|
| 11 |
+
# ============================================================================
|
| 12 |
+
# Recommended hardware:
|
| 13 |
+
# - 8× H100 80GB (TP=8) or 16× A100 80GB (TP=8, EP=2)
|
| 14 |
+
# - ~600 GB RAM for data loading
|
| 15 |
+
# - 3M context requires gradient checkpointing + KV int8 cache
|
| 16 |
+
# ============================================================================
|
| 17 |
+
|
| 18 |
+
name: "Nexus Coder 423B"
|
| 19 |
+
version: "0.4.0"
|
| 20 |
+
author: "Hieu Louis"
|
| 21 |
+
|
| 22 |
+
# === Architecture ===
|
| 23 |
+
vocab_size: 200000
|
| 24 |
+
hidden_size: 7168
|
| 25 |
+
num_hidden_layers: 24
|
| 26 |
+
num_attention_heads: 56
|
| 27 |
+
num_kv_heads: 8
|
| 28 |
+
head_dim: 128
|
| 29 |
+
intermediate_size: 16384
|
| 30 |
+
hidden_act: "silu"
|
| 31 |
+
num_experts: 48
|
| 32 |
+
num_active_experts: 4
|
| 33 |
+
router_aux_loss_coef: 0.001
|
| 34 |
+
|
| 35 |
+
# === Context window (3M tokens via YaRN ×60) ===
|
| 36 |
+
max_position_embeddings: 3000000
|
| 37 |
+
rotary_emb_base: 1000000.0 # larger base for long context
|
| 38 |
+
rope_scaling_type: "yarn"
|
| 39 |
+
rope_scaling_factor: 60.0
|
| 40 |
+
yarn_beta_fast: 32.0
|
| 41 |
+
yarn_beta_slow: 1.0
|
| 42 |
+
|
| 43 |
+
# === Attention features ===
|
| 44 |
+
use_flash_attention: true
|
| 45 |
+
use_flash_attention_2: true
|
| 46 |
+
use_qk_norm: true
|
| 47 |
+
qk_norm_eps: 1.0e-6
|
| 48 |
+
mlp_parallel: true
|
| 49 |
+
use_sliding_window: true
|
| 50 |
+
sliding_window_size: 32768
|
| 51 |
+
use_alibi: false
|
| 52 |
+
|
| 53 |
+
# === Memory optimizations ===
|
| 54 |
+
gradient_checkpointing: true
|
| 55 |
+
kv_cache_quantization: "int8"
|
| 56 |
+
kv_cache_bits: 8
|
| 57 |
+
|
| 58 |
+
# === v0.4 CyberGym ===
|
| 59 |
+
cybergym_enabled: true
|
| 60 |
+
cybergym_mutation_rate: 0.01
|
| 61 |
+
cybergym_mutation_sigma: 1.0e-4
|
| 62 |
+
cybergym_mutation_period: 500
|
| 63 |
+
cybergym_keep_ratio: 0.7
|
| 64 |
+
cybergym_adaptive_routing: true
|
| 65 |
+
cybergym_min_active_experts: 2
|
| 66 |
+
cybergym_max_active_experts: 8
|
| 67 |
+
cybergym_genome_init: true
|
| 68 |
+
cybergym_cep_stages: [32768, 131072, 524288, 1048576, 2097152, 3000000]
|
| 69 |
+
cybergym_cep_epoch_per_stage: 1
|
| 70 |
+
|
| 71 |
+
# === Distributed ===
|
| 72 |
+
tensor_parallel_size: 8
|
| 73 |
+
pipeline_parallel_size: 1
|
| 74 |
+
expert_parallel_size: 8
|
| 75 |
+
sequence_parallel: false
|
| 76 |
+
|
| 77 |
+
# === Training defaults ===
|
| 78 |
+
pad_token_id: 0
|
| 79 |
+
bos_token_id: 1
|
| 80 |
+
eos_token_id: 2
|
| 81 |
+
unk_token_id: 3
|
| 82 |
+
|
| 83 |
+
# === Safety ===
|
| 84 |
+
enable_safety_filter: true
|
| 85 |
+
max_output_tokens: 8192
|
| 86 |
+
|
| 87 |
+
# === Personality ===
|
| 88 |
+
personality: "humorous"
|
| 89 |
+
language: "bilingual"
|
configs/nexus_coder_70b.yaml
ADDED
|
@@ -0,0 +1,106 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Nexus Coder Configuration - 70B/5B v0.3 NEW (frontier research)
|
| 2 |
+
# Author: Hieu Louis (2026)
|
| 3 |
+
# RESEARCH ONLY. Requires 256+ H100/H200 GPUs or equivalent.
|
| 4 |
+
# Uses YaRN RoPE scaling for 4× context extension → 128k tokens.
|
| 5 |
+
|
| 6 |
+
model:
|
| 7 |
+
name: "Nexus Coder 70B"
|
| 8 |
+
agent_name: "Nexus"
|
| 9 |
+
author: "Hieu Louis"
|
| 10 |
+
version: "0.3.0-70b"
|
| 11 |
+
github: "mhieuhonda"
|
| 12 |
+
year: "2026"
|
| 13 |
+
|
| 14 |
+
architecture:
|
| 15 |
+
vocab_size: 128000 # tiktoken-style tokenizer
|
| 16 |
+
hidden_size: 6144
|
| 17 |
+
num_hidden_layers: 32
|
| 18 |
+
num_attention_heads: 48
|
| 19 |
+
num_kv_heads: 8 # heavy GQA (6:1 ratio)
|
| 20 |
+
head_dim: 128
|
| 21 |
+
intermediate_size: 16384
|
| 22 |
+
hidden_act: "silu"
|
| 23 |
+
norm_type: "rmsnorm"
|
| 24 |
+
|
| 25 |
+
moe:
|
| 26 |
+
num_experts: 64 # Frontier-scale MoE
|
| 27 |
+
num_active_experts: 4
|
| 28 |
+
router_aux_loss_coef: 0.001
|
| 29 |
+
router_jitter_noise: 0.0
|
| 30 |
+
|
| 31 |
+
context:
|
| 32 |
+
max_position_embeddings: 131072 # 128k tokens
|
| 33 |
+
rotary_emb_base: 500000.0 # larger base for long context
|
| 34 |
+
rope_scaling_type: "yarn" # YaRN — SOTA for 4×+ extension
|
| 35 |
+
rope_scaling_factor: 4.0
|
| 36 |
+
yarn_beta_fast: 32.0
|
| 37 |
+
yarn_beta_slow: 1.0
|
| 38 |
+
|
| 39 |
+
attention:
|
| 40 |
+
use_flash_attention: true
|
| 41 |
+
use_flash_attention_2: true
|
| 42 |
+
use_alibi: false # YaRN handles long context
|
| 43 |
+
use_sliding_window: true
|
| 44 |
+
sliding_window_size: 16384
|
| 45 |
+
use_qk_norm: true
|
| 46 |
+
qk_norm_eps: 1.0e-6
|
| 47 |
+
mlp_parallel: true
|
| 48 |
+
|
| 49 |
+
compute:
|
| 50 |
+
use_kv_cache: true
|
| 51 |
+
kv_cache_quantization: "fp8" # FP8 KV cache for memory efficiency
|
| 52 |
+
gradient_checkpointing: true
|
| 53 |
+
tensor_parallel_size: 8
|
| 54 |
+
pipeline_parallel_size: 2
|
| 55 |
+
expert_parallel_size: 8
|
| 56 |
+
sequence_parallel: true # enable sequence parallel for long context
|
| 57 |
+
|
| 58 |
+
params:
|
| 59 |
+
total: "~70B"
|
| 60 |
+
active: "~5B"
|
| 61 |
+
expert_utilization: "6.25%"
|
| 62 |
+
estimated_disk_mb_fp16: 140000
|
| 63 |
+
estimated_disk_mb_int8: 70000
|
| 64 |
+
estimated_disk_mb_int4: 35000
|
| 65 |
+
kv_cache_mb_per_token_fp16: 0.050
|
| 66 |
+
kv_cache_mb_per_token_fp8: 0.025
|
| 67 |
+
|
| 68 |
+
training:
|
| 69 |
+
learning_rate: 1.5e-4
|
| 70 |
+
weight_decay: 0.01
|
| 71 |
+
warmup_steps: 2000
|
| 72 |
+
max_steps: 50000
|
| 73 |
+
per_device_batch_size: 1
|
| 74 |
+
gradient_accumulation_steps: 128
|
| 75 |
+
logging_steps: 10
|
| 76 |
+
save_steps: 2000
|
| 77 |
+
max_grad_norm: 1.0
|
| 78 |
+
seed: 42
|
| 79 |
+
use_amp: true
|
| 80 |
+
use_deepspeed: true
|
| 81 |
+
deepspeed_config: "configs/ds_config_zero3_offload.json"
|
| 82 |
+
total_tokens_target: 1_500_000_000_000 # 1.5T tokens
|
| 83 |
+
|
| 84 |
+
inference:
|
| 85 |
+
max_new_tokens: 2000
|
| 86 |
+
temperature: 0.7
|
| 87 |
+
top_k: 50
|
| 88 |
+
top_p: 0.9
|
| 89 |
+
do_sample: true
|
| 90 |
+
|
| 91 |
+
personality:
|
| 92 |
+
type: "humorous"
|
| 93 |
+
language: "bilingual"
|
| 94 |
+
|
| 95 |
+
capabilities:
|
| 96 |
+
skills_count: 60
|
| 97 |
+
tools_count: 80
|
| 98 |
+
|
| 99 |
+
environment:
|
| 100 |
+
python_version: "3.12.13"
|
| 101 |
+
pytorch_version: ">=2.3"
|
| 102 |
+
cuda_required: true
|
| 103 |
+
min_gpu_memory_gb: 80
|
| 104 |
+
recommended_gpus: "256+ H100/H200 80GB"
|
| 105 |
+
estimated_training_time: "~90 days on 256 H100s"
|
| 106 |
+
notes: "This config is research-only. Use 30B or 10B for production."
|
configs/nexus_coder_medium.yaml
ADDED
|
@@ -0,0 +1,80 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Nexus Coder Configuration - Medium version v0.3
|
| 2 |
+
# ~1B params, pretrain on 4-8 GPU
|
| 3 |
+
# Author: Hieu Louis (2026)
|
| 4 |
+
|
| 5 |
+
model:
|
| 6 |
+
name: "Nexus Coder Medium"
|
| 7 |
+
agent_name: "Nexus"
|
| 8 |
+
author: "Hieu Louis"
|
| 9 |
+
version: "0.3.0-medium"
|
| 10 |
+
github: "mhieuhonda"
|
| 11 |
+
year: "2026"
|
| 12 |
+
|
| 13 |
+
architecture:
|
| 14 |
+
vocab_size: 32000
|
| 15 |
+
hidden_size: 1536
|
| 16 |
+
num_hidden_layers: 24
|
| 17 |
+
num_attention_heads: 16
|
| 18 |
+
num_kv_heads: 4
|
| 19 |
+
head_dim: 96
|
| 20 |
+
intermediate_size: 4096
|
| 21 |
+
hidden_act: "silu"
|
| 22 |
+
norm_type: "rmsnorm"
|
| 23 |
+
|
| 24 |
+
moe:
|
| 25 |
+
num_experts: 16
|
| 26 |
+
num_active_experts: 2
|
| 27 |
+
router_aux_loss_coef: 0.001
|
| 28 |
+
|
| 29 |
+
context:
|
| 30 |
+
max_position_embeddings: 16384
|
| 31 |
+
rotary_emb_base: 10000.0
|
| 32 |
+
|
| 33 |
+
attention:
|
| 34 |
+
use_flash_attention: true
|
| 35 |
+
use_flash_attention_2: false
|
| 36 |
+
use_alibi: false
|
| 37 |
+
use_sliding_window: true
|
| 38 |
+
sliding_window_size: 2048
|
| 39 |
+
use_qk_norm: true
|
| 40 |
+
mlp_parallel: true
|
| 41 |
+
|
| 42 |
+
compute:
|
| 43 |
+
use_kv_cache: true
|
| 44 |
+
kv_cache_quantization: null
|
| 45 |
+
gradient_checkpointing: false
|
| 46 |
+
|
| 47 |
+
params:
|
| 48 |
+
total: "~1.1B"
|
| 49 |
+
active: "~250M"
|
| 50 |
+
expert_utilization: "12.5%"
|
| 51 |
+
|
| 52 |
+
training:
|
| 53 |
+
learning_rate: 3.0e-4
|
| 54 |
+
weight_decay: 0.01
|
| 55 |
+
warmup_steps: 100
|
| 56 |
+
max_steps: 5000
|
| 57 |
+
per_device_batch_size: 4
|
| 58 |
+
gradient_accumulation_steps: 4
|
| 59 |
+
logging_steps: 10
|
| 60 |
+
save_steps: 500
|
| 61 |
+
max_grad_norm: 1.0
|
| 62 |
+
seed: 42
|
| 63 |
+
use_amp: true
|
| 64 |
+
|
| 65 |
+
inference:
|
| 66 |
+
max_new_tokens: 200
|
| 67 |
+
temperature: 0.8
|
| 68 |
+
top_k: 50
|
| 69 |
+
top_p: 0.9
|
| 70 |
+
do_sample: true
|
| 71 |
+
|
| 72 |
+
personality:
|
| 73 |
+
type: "humorous"
|
| 74 |
+
language: "bilingual"
|
| 75 |
+
|
| 76 |
+
environment:
|
| 77 |
+
python_version: "3.12.13"
|
| 78 |
+
pytorch_version: ">=2.0"
|
| 79 |
+
cuda_required: true
|
| 80 |
+
min_gpu_memory_gb: 16
|
configs/nexus_coder_small.yaml
ADDED
|
@@ -0,0 +1,77 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Nexus Coder Configuration - Small version v0.3
|
| 2 |
+
# ~125M params, fine-tune on 1 GPU
|
| 3 |
+
# Author: Hieu Louis (2026)
|
| 4 |
+
|
| 5 |
+
model:
|
| 6 |
+
name: "Nexus Coder Small"
|
| 7 |
+
agent_name: "Nexus"
|
| 8 |
+
author: "Hieu Louis"
|
| 9 |
+
version: "0.3.0-small"
|
| 10 |
+
github: "mhieuhonda"
|
| 11 |
+
year: "2026"
|
| 12 |
+
|
| 13 |
+
architecture:
|
| 14 |
+
vocab_size: 16000
|
| 15 |
+
hidden_size: 768
|
| 16 |
+
num_hidden_layers: 12
|
| 17 |
+
num_attention_heads: 12
|
| 18 |
+
num_kv_heads: 4
|
| 19 |
+
head_dim: 64
|
| 20 |
+
intermediate_size: 2048
|
| 21 |
+
hidden_act: "silu"
|
| 22 |
+
norm_type: "rmsnorm"
|
| 23 |
+
|
| 24 |
+
moe:
|
| 25 |
+
num_experts: 8
|
| 26 |
+
num_active_experts: 2
|
| 27 |
+
router_aux_loss_coef: 0.001
|
| 28 |
+
|
| 29 |
+
context:
|
| 30 |
+
max_position_embeddings: 8192
|
| 31 |
+
rotary_emb_base: 10000.0
|
| 32 |
+
|
| 33 |
+
attention:
|
| 34 |
+
use_flash_attention: true
|
| 35 |
+
use_flash_attention_2: false
|
| 36 |
+
use_alibi: false
|
| 37 |
+
use_sliding_window: false
|
| 38 |
+
use_qk_norm: true
|
| 39 |
+
mlp_parallel: true
|
| 40 |
+
|
| 41 |
+
compute:
|
| 42 |
+
use_kv_cache: true
|
| 43 |
+
kv_cache_quantization: null
|
| 44 |
+
gradient_checkpointing: false
|
| 45 |
+
|
| 46 |
+
params:
|
| 47 |
+
total: "~125M"
|
| 48 |
+
active: "~45M"
|
| 49 |
+
expert_utilization: "25%"
|
| 50 |
+
|
| 51 |
+
training:
|
| 52 |
+
learning_rate: 3.0e-4
|
| 53 |
+
weight_decay: 0.01
|
| 54 |
+
warmup_steps: 50
|
| 55 |
+
max_steps: 1000
|
| 56 |
+
per_device_batch_size: 8
|
| 57 |
+
gradient_accumulation_steps: 2
|
| 58 |
+
logging_steps: 10
|
| 59 |
+
save_steps: 200
|
| 60 |
+
max_grad_norm: 1.0
|
| 61 |
+
seed: 42
|
| 62 |
+
|
| 63 |
+
inference:
|
| 64 |
+
max_new_tokens: 200
|
| 65 |
+
temperature: 0.8
|
| 66 |
+
top_k: 50
|
| 67 |
+
top_p: 0.9
|
| 68 |
+
do_sample: true
|
| 69 |
+
|
| 70 |
+
personality:
|
| 71 |
+
type: "humorous"
|
| 72 |
+
language: "bilingual"
|
| 73 |
+
|
| 74 |
+
environment:
|
| 75 |
+
python_version: "3.12.13"
|
| 76 |
+
pytorch_version: ">=2.0"
|
| 77 |
+
cuda_required: false
|
configs/nexus_coder_tiny.yaml
ADDED
|
@@ -0,0 +1,80 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Nexus Coder Configuration - Tiny version v0.3
|
| 2 |
+
# Used for quick verification on CPU (~5M params)
|
| 3 |
+
# Author: Hieu Louis (2026)
|
| 4 |
+
|
| 5 |
+
model:
|
| 6 |
+
name: "Nexus Coder Tiny"
|
| 7 |
+
agent_name: "Nexus"
|
| 8 |
+
author: "Hieu Louis"
|
| 9 |
+
version: "0.3.0-tiny"
|
| 10 |
+
github: "mhieuhonda"
|
| 11 |
+
year: "2026"
|
| 12 |
+
|
| 13 |
+
architecture:
|
| 14 |
+
vocab_size: 2000
|
| 15 |
+
hidden_size: 256
|
| 16 |
+
num_hidden_layers: 4
|
| 17 |
+
num_attention_heads: 8
|
| 18 |
+
num_kv_heads: 2
|
| 19 |
+
head_dim: 32
|
| 20 |
+
intermediate_size: 512
|
| 21 |
+
hidden_act: "silu"
|
| 22 |
+
norm_type: "rmsnorm"
|
| 23 |
+
|
| 24 |
+
moe:
|
| 25 |
+
num_experts: 4
|
| 26 |
+
num_active_experts: 2
|
| 27 |
+
router_aux_loss_coef: 0.001
|
| 28 |
+
router_jitter_noise: 0.0
|
| 29 |
+
|
| 30 |
+
context:
|
| 31 |
+
max_position_embeddings: 512
|
| 32 |
+
rotary_emb_base: 10000.0
|
| 33 |
+
rope_scaling_type: null
|
| 34 |
+
rope_scaling_factor: 1.0
|
| 35 |
+
|
| 36 |
+
# v0.3 NEW architecture features (most OFF for tiny — too small to benefit)
|
| 37 |
+
attention:
|
| 38 |
+
use_flash_attention: false
|
| 39 |
+
use_flash_attention_2: false
|
| 40 |
+
use_alibi: false
|
| 41 |
+
use_sliding_window: false
|
| 42 |
+
sliding_window_size: 256
|
| 43 |
+
use_qk_norm: false
|
| 44 |
+
mlp_parallel: true
|
| 45 |
+
|
| 46 |
+
compute:
|
| 47 |
+
use_kv_cache: true
|
| 48 |
+
kv_cache_quantization: null
|
| 49 |
+
gradient_checkpointing: false
|
| 50 |
+
|
| 51 |
+
params:
|
| 52 |
+
total: "~8M (demo only)"
|
| 53 |
+
active: "~5M"
|
| 54 |
+
note: "For testing only. Use nexus_coder_10b.yaml for the real model."
|
| 55 |
+
|
| 56 |
+
training:
|
| 57 |
+
learning_rate: 5.0e-4
|
| 58 |
+
weight_decay: 0.01
|
| 59 |
+
warmup_steps: 10
|
| 60 |
+
max_steps: 30
|
| 61 |
+
per_device_batch_size: 2
|
| 62 |
+
gradient_accumulation_steps: 1
|
| 63 |
+
logging_steps: 5
|
| 64 |
+
save_steps: 30
|
| 65 |
+
|
| 66 |
+
inference:
|
| 67 |
+
max_new_tokens: 50
|
| 68 |
+
temperature: 0.8
|
| 69 |
+
top_k: 50
|
| 70 |
+
top_p: 0.9
|
| 71 |
+
do_sample: true
|
| 72 |
+
|
| 73 |
+
personality:
|
| 74 |
+
type: "humorous"
|
| 75 |
+
language: "bilingual"
|
| 76 |
+
|
| 77 |
+
environment:
|
| 78 |
+
python_version: "3.12.13"
|
| 79 |
+
pytorch_version: ">=2.0"
|
| 80 |
+
cuda_required: false
|
configs/nexus_coder_xlarge.yaml
ADDED
|
@@ -0,0 +1,92 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Nexus Coder Configuration - XLarge (~30B/3B) v0.3
|
| 2 |
+
# Research-only. Requires 64+ H100 80GB GPUs.
|
| 3 |
+
# Author: Hieu Louis (2026)
|
| 4 |
+
|
| 5 |
+
model:
|
| 6 |
+
name: "Nexus Coder XLarge"
|
| 7 |
+
agent_name: "Nexus"
|
| 8 |
+
author: "Hieu Louis"
|
| 9 |
+
version: "0.3.0-xlarge"
|
| 10 |
+
github: "mhieuhonda"
|
| 11 |
+
year: "2026"
|
| 12 |
+
|
| 13 |
+
architecture:
|
| 14 |
+
vocab_size: 64000
|
| 15 |
+
hidden_size: 4096
|
| 16 |
+
num_hidden_layers: 24
|
| 17 |
+
num_attention_heads: 32
|
| 18 |
+
num_kv_heads: 8
|
| 19 |
+
head_dim: 128
|
| 20 |
+
intermediate_size: 11264
|
| 21 |
+
hidden_act: "silu"
|
| 22 |
+
norm_type: "rmsnorm"
|
| 23 |
+
|
| 24 |
+
moe:
|
| 25 |
+
num_experts: 48
|
| 26 |
+
num_active_experts: 4
|
| 27 |
+
router_aux_loss_coef: 0.001
|
| 28 |
+
|
| 29 |
+
context:
|
| 30 |
+
max_position_embeddings: 65536 # 64k tokens
|
| 31 |
+
rotary_emb_base: 10000.0
|
| 32 |
+
rope_scaling_type: "dynamic" # NTK-aware for 2× context extension
|
| 33 |
+
rope_scaling_factor: 2.0
|
| 34 |
+
|
| 35 |
+
attention:
|
| 36 |
+
use_flash_attention: true
|
| 37 |
+
use_flash_attention_2: true # recommended at this scale
|
| 38 |
+
use_alibi: false
|
| 39 |
+
use_sliding_window: true
|
| 40 |
+
sliding_window_size: 8192
|
| 41 |
+
use_qk_norm: true
|
| 42 |
+
mlp_parallel: true
|
| 43 |
+
|
| 44 |
+
compute:
|
| 45 |
+
use_kv_cache: true
|
| 46 |
+
kv_cache_quantization: "int8" # saves KV cache memory at long context
|
| 47 |
+
gradient_checkpointing: true # essential at this scale
|
| 48 |
+
tensor_parallel_size: 4
|
| 49 |
+
pipeline_parallel_size: 1
|
| 50 |
+
expert_parallel_size: 4
|
| 51 |
+
sequence_parallel: false
|
| 52 |
+
|
| 53 |
+
params:
|
| 54 |
+
total: "~30B"
|
| 55 |
+
active: "~3B"
|
| 56 |
+
expert_utilization: "8.3%"
|
| 57 |
+
estimated_disk_mb_fp16: 60000
|
| 58 |
+
estimated_disk_mb_int8: 30000
|
| 59 |
+
estimated_disk_mb_int4: 15000
|
| 60 |
+
|
| 61 |
+
training:
|
| 62 |
+
learning_rate: 2.0e-4
|
| 63 |
+
weight_decay: 0.01
|
| 64 |
+
warmup_steps: 500
|
| 65 |
+
max_steps: 10000
|
| 66 |
+
per_device_batch_size: 1
|
| 67 |
+
gradient_accumulation_steps: 32
|
| 68 |
+
logging_steps: 10
|
| 69 |
+
save_steps: 1000
|
| 70 |
+
max_grad_norm: 1.0
|
| 71 |
+
seed: 42
|
| 72 |
+
use_amp: true
|
| 73 |
+
use_deepspeed: true
|
| 74 |
+
deepspeed_config: "configs/ds_config_zero3.json"
|
| 75 |
+
|
| 76 |
+
inference:
|
| 77 |
+
max_new_tokens: 500
|
| 78 |
+
temperature: 0.7
|
| 79 |
+
top_k: 50
|
| 80 |
+
top_p: 0.9
|
| 81 |
+
do_sample: true
|
| 82 |
+
|
| 83 |
+
personality:
|
| 84 |
+
type: "humorous"
|
| 85 |
+
language: "bilingual"
|
| 86 |
+
|
| 87 |
+
environment:
|
| 88 |
+
python_version: "3.12.13"
|
| 89 |
+
pytorch_version: ">=2.0"
|
| 90 |
+
cuda_required: true
|
| 91 |
+
min_gpu_memory_gb: 80
|
| 92 |
+
recommended_gpus: "64+ H100 80GB"
|
configs/sources.yaml
ADDED
|
@@ -0,0 +1,685 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Nexus Coder v0.3 - Training Data Sources Configuration
|
| 2 |
+
# =======================================================
|
| 3 |
+
# Curated sources for pre-training Nexus Coder v0.3.
|
| 4 |
+
# Target: ~500 GitHub repos + ~150 HuggingFace datasets + 5 new sources.
|
| 5 |
+
#
|
| 6 |
+
# Author: Hieu Louis (2026)
|
| 7 |
+
# Total estimated tokens (post-filtering): ~50B-200B
|
| 8 |
+
#
|
| 9 |
+
# References (the inspiration for many of these sources):
|
| 10 |
+
# - litgpt's curated pretraining datasets
|
| 11 |
+
# - LlamaFactory's example configs
|
| 12 |
+
# - axolotl's dataset registry
|
| 13 |
+
# - StarCoder2 paper data card
|
| 14 |
+
# - The-Stack v2 dataset card
|
| 15 |
+
|
| 16 |
+
# =============================================================================
|
| 17 |
+
# GitHub — curated code repos (~500)
|
| 18 |
+
# =============================================================================
|
| 19 |
+
github:
|
| 20 |
+
enabled: true
|
| 21 |
+
cache_dir: "./data_cache/github"
|
| 22 |
+
max_concurrent: 8
|
| 23 |
+
max_files_per_repo: 1000
|
| 24 |
+
max_file_size_kb: 100
|
| 25 |
+
|
| 26 |
+
repos:
|
| 27 |
+
# ---------- Python core & stdlib (5) ----------
|
| 28 |
+
- {owner: "python", name: "cpython", languages: ["python"], max_files: 3000}
|
| 29 |
+
- {owner: "pallets", name: "flask", languages: ["python"]}
|
| 30 |
+
- {owner: "pallets", name: "django", languages: ["python"], max_files: 2000}
|
| 31 |
+
- {owner: "psf", name: "requests", languages: ["python"]}
|
| 32 |
+
- {owner: "pallets", name: "click", languages: ["python"]}
|
| 33 |
+
|
| 34 |
+
# ---------- Python: data science (8) ----------
|
| 35 |
+
- {owner: "numpy", name: "numpy", languages: ["python"], max_files: 2000}
|
| 36 |
+
- {owner: "pandas-dev", name: "pandas", languages: ["python"], max_files: 2000}
|
| 37 |
+
- {owner: "scipy", name: "scipy", languages: ["python"], max_files: 2000}
|
| 38 |
+
- {owner: "matplotlib", name: "matplotlib", languages: ["python"], max_files: 2000}
|
| 39 |
+
- {owner: "scikit-learn", name: "scikit-learn", languages: ["python"], max_files: 2000}
|
| 40 |
+
- {owner: "plotly", name: "plotly.py", languages: ["python"]}
|
| 41 |
+
- {owner: "bokeh", name: "bokeh", languages: ["python"]}
|
| 42 |
+
- {owner: "sympy", name: "sympy", languages: ["python"], max_files: 2000}
|
| 43 |
+
|
| 44 |
+
# ---------- Python: ML / DL (10) ----------
|
| 45 |
+
- {owner: "pytorch", name: "pytorch", languages: ["python", "cpp"], max_files: 3000}
|
| 46 |
+
- {owner: "tensorflow", name: "tensorflow", languages: ["python", "cpp"], max_files: 3000}
|
| 47 |
+
- {owner: "huggingface", name: "transformers", languages: ["python"], max_files: 3000}
|
| 48 |
+
- {owner: "huggingface", name: "datasets", languages: ["python"]}
|
| 49 |
+
- {owner: "huggingface", name: "peft", languages: ["python"]}
|
| 50 |
+
- {owner: "huggingface", name: "accelerate", languages: ["python"]}
|
| 51 |
+
- {owner: "huggingface", name: "tokenizers", languages: ["python", "rust"]}
|
| 52 |
+
- {owner: "langchain-ai", name: "langchain", languages: ["python"], max_files: 2000}
|
| 53 |
+
- {owner: "run-llama", name: "llama_index", languages: ["python"]}
|
| 54 |
+
- {owner: "explosion", name: "spaCy", languages: ["python"]}
|
| 55 |
+
|
| 56 |
+
# ---------- Python: Web frameworks (8) ----------
|
| 57 |
+
- {owner: "tiangolo", name: "fastapi", languages: ["python"], max_files: 2000}
|
| 58 |
+
- {owner: "encode", name: "starlette", languages: ["python"]}
|
| 59 |
+
- {owner: "encode", name: "uvicorn", languages: ["python"]}
|
| 60 |
+
- {owner: "django", name: "djangoproject.com", languages: ["python"]}
|
| 61 |
+
- {owner: "falconry", name: "falcon", languages: ["python"]}
|
| 62 |
+
- {owner: "sanic-org", name: "sanic", languages: ["python"]}
|
| 63 |
+
- {owner: "tornadoweb", name: "tornado", languages: ["python"]}
|
| 64 |
+
- {owner: "aio-libs", name: "aiohttp", languages: ["python"]}
|
| 65 |
+
|
| 66 |
+
# ---------- Python: tools (8) ----------
|
| 67 |
+
- {owner: "pytest-dev", name: "pytest", languages: ["python"]}
|
| 68 |
+
- {owner: "psf", name: "black", languages: ["python"]}
|
| 69 |
+
- {owner: "pydantic", name: "pydantic", languages: ["python"]}
|
| 70 |
+
- {owner: "pypa", name: "pip", languages: ["python"]}
|
| 71 |
+
- {owner: "pypa", name: "setuptools", languages: ["python"]}
|
| 72 |
+
- {owner: "pyca", name: "cryptography", languages: ["python", "c"]}
|
| 73 |
+
- {owner: "celery", name: "celery", languages: ["python"]}
|
| 74 |
+
- {owner: "mwclient", name: "redis-py", languages: ["python"]}
|
| 75 |
+
|
| 76 |
+
# ---------- Python: async / networking (5) ----------
|
| 77 |
+
- {owner: "aio-libs", name: "aiomysql", languages: ["python"]}
|
| 78 |
+
- {owner: "MagicStack", name: "asyncpg", languages: ["python", "cython"]}
|
| 79 |
+
- {owner: "sqlalchemy", name: "sqlalchemy", languages: ["python"], max_files: 2000}
|
| 80 |
+
- {owner: "scrapy", name: "scrapy", languages: ["python"]}
|
| 81 |
+
- {owner: "httpx", name: "httpx", languages: ["python"]}
|
| 82 |
+
|
| 83 |
+
# ---------- Python: DevOps / Infra (5) ----------
|
| 84 |
+
- {owner: "ansible", name: "ansible", languages: ["python"], max_files: 2000}
|
| 85 |
+
- {owner: "openstack", name: "openstack", languages: ["python"]}
|
| 86 |
+
- {owner: "saltstack", name: "salt", languages: ["python"]}
|
| 87 |
+
- {owner: "aws", name: "aws-cli", languages: ["python"]}
|
| 88 |
+
- {owner: "boto", name: "boto3", languages: ["python"]}
|
| 89 |
+
|
| 90 |
+
# ---------- JavaScript / TypeScript (10) ----------
|
| 91 |
+
- {owner: "facebook", name: "react", languages: ["javascript", "typescript"], max_files: 2000}
|
| 92 |
+
- {owner: "vuejs", name: "vue", languages: ["javascript", "typescript"], max_files: 2000}
|
| 93 |
+
- {owner: "vercel", name: "next.js", languages: ["javascript", "typescript"], max_files: 2000}
|
| 94 |
+
- {owner: "angular", name: "angular", languages: ["typescript"], max_files: 2000}
|
| 95 |
+
- {owner: "sveltejs", name: "svelte", languages: ["javascript", "typescript"]}
|
| 96 |
+
- {owner: "microsoft", name: "TypeScript", languages: ["typescript"], max_files: 3000}
|
| 97 |
+
- {owner: "nodejs", name: "node", languages: ["javascript", "cpp"], max_files: 2000}
|
| 98 |
+
- {owner: "denoland", name: "deno", languages: ["typescript", "rust"], max_files: 2000}
|
| 99 |
+
- {owner: "expressjs", name: "express", languages: ["javascript"]}
|
| 100 |
+
- {owner: "fastify", name: "fastify", languages: ["javascript"]}
|
| 101 |
+
|
| 102 |
+
# ---------- JavaScript / TypeScript: tools (5) ----------
|
| 103 |
+
- {owner: "eslint", name: "eslint", languages: ["javascript"]}
|
| 104 |
+
- {owner: "prettier", name: "prettier", languages: ["javascript", "typescript"]}
|
| 105 |
+
- {owner: "webpack", name: "webpack", languages: ["javascript"], max_files: 2000}
|
| 106 |
+
- {owner: "vitejs", name: "vite", languages: ["typescript"]}
|
| 107 |
+
- {owner: "rollup", name: "rollup", languages: ["javascript", "typescript"]}
|
| 108 |
+
|
| 109 |
+
# ---------- Go (10) ----------
|
| 110 |
+
- {owner: "golang", name: "go", languages: ["go"], max_files: 3000}
|
| 111 |
+
- {owner: "gin-gonic", name: "gin", languages: ["go"]}
|
| 112 |
+
- {owner: "kubernetes", name: "kubernetes", languages: ["go"], max_files: 3000}
|
| 113 |
+
- {owner: "prometheus", name: "prometheus", languages: ["go"], max_files: 2000}
|
| 114 |
+
- {owner: "hashicorp", name: "terraform", languages: ["go"], max_files: 2000}
|
| 115 |
+
- {owner: "hashicorp", name: "consul", languages: ["go"]}
|
| 116 |
+
- {owner: "hashicorp", name: "vault", languages: ["go"]}
|
| 117 |
+
- {owner: "etcd-io", name: "etcd", languages: ["go"]}
|
| 118 |
+
- {owner: "docker", name: "compose", languages: ["go"]}
|
| 119 |
+
- {owner: "gohugoio", name: "hugo", languages: ["go"]}
|
| 120 |
+
|
| 121 |
+
# ---------- Go: more tools (5) ----------
|
| 122 |
+
- {owner: "spf13", name: "cobra", languages: ["go"]}
|
| 123 |
+
- {owner: "spf13", name: "viper", languages: ["go"]}
|
| 124 |
+
- {owner: "golang", name: "mock", languages: ["go"]}
|
| 125 |
+
- {owner: "stretchr", name: "testify", languages: ["go"]}
|
| 126 |
+
- {owner: "grpc", name: "grpc-go", languages: ["go"]}
|
| 127 |
+
|
| 128 |
+
# ---------- Rust (10) ----------
|
| 129 |
+
- {owner: "rust-lang", name: "rust", languages: ["rust"], max_files: 3000}
|
| 130 |
+
- {owner: "tokio-rs", name: "tokio", languages: ["rust"], max_files: 2000}
|
| 131 |
+
- {owner: "serde-rs", name: "serde", languages: ["rust"]}
|
| 132 |
+
- {owner: "BurntSushi", name: "ripgrep", languages: ["rust"]}
|
| 133 |
+
- {owner: "sharkdp", name: "bat", languages: ["rust"]}
|
| 134 |
+
- {owner: "sharkdp", name: "fd", languages: ["rust"]}
|
| 135 |
+
- {owner: "BurntSushi", name: "csv", languages: ["rust"]}
|
| 136 |
+
- {owner: "rust-lang", name: "cargo", languages: ["rust"]}
|
| 137 |
+
- {owner: "rust-lang", name: "rustfmt", languages: ["rust"]}
|
| 138 |
+
- {owner: "delta-io", name: "delta-rs", languages: ["rust"]}
|
| 139 |
+
|
| 140 |
+
# ---------- Rust: web / async (5) ----------
|
| 141 |
+
- {owner: "actix", name: "actix-web", languages: ["rust"]}
|
| 142 |
+
- {owner: "axo", name: "axum", languages: ["rust"]}
|
| 143 |
+
- {owner: "hyperium", name: "hyper", languages: ["rust"]}
|
| 144 |
+
- {owner: "hyperium", name: "tonic", languages: ["rust"]}
|
| 145 |
+
- {owner: "seanmonstar", name: "reqwest", languages: ["rust"]}
|
| 146 |
+
|
| 147 |
+
# ---------- C / C++ (8) ----------
|
| 148 |
+
- {owner: "llvm", name: "llvm-project", languages: ["cpp"], max_files: 3000}
|
| 149 |
+
- {owner: "gcc-mirror", name: "gcc", languages: ["cpp", "c"], max_files: 2000}
|
| 150 |
+
- {owner: "cmake", name: "cmake", languages: ["cpp"]}
|
| 151 |
+
- {owner: "google", name: "googletest", languages: ["cpp"]}
|
| 152 |
+
- {owner: "fmtlib", name: "fmt", languages: ["cpp"]}
|
| 153 |
+
- {owner: "gabime", name: "spdlog", languages: ["cpp"]}
|
| 154 |
+
- {owner: "nlohmann", name: "json", languages: ["cpp"]}
|
| 155 |
+
- {owner: "grpc", name: "grpc", languages: ["cpp", "c"], max_files: 2000}
|
| 156 |
+
|
| 157 |
+
# ---------- Java (6) ----------
|
| 158 |
+
- {owner: "spring-projects", name: "spring-boot", languages: ["java"], max_files: 2000}
|
| 159 |
+
- {owner: "apache", name: "kafka", languages: ["java", "scala"], max_files: 2000}
|
| 160 |
+
- {owner: "apache", name: "cassandra", languages: ["java"]}
|
| 161 |
+
- {owner: "apache", name: "maven", languages: ["java"]}
|
| 162 |
+
- {owner: "apache", name: "tomcat", languages: ["java"]}
|
| 163 |
+
- {owner: "OpenLiberty", name: "open-liberty", languages: ["java"]}
|
| 164 |
+
|
| 165 |
+
# ---------- Java: tools (4) ----------
|
| 166 |
+
- {owner: "junit-team", name: "junit5", languages: ["java"]}
|
| 167 |
+
- {owner: "mockito", name: "mockito", languages: ["java"]}
|
| 168 |
+
- {owner: "GoogleJavaFormat", name: "google-java-format", languages: ["java"]}
|
| 169 |
+
- {owner: "checkstyle", name: "checkstyle", languages: ["java"]}
|
| 170 |
+
|
| 171 |
+
# ---------- C# / .NET (4) ----------
|
| 172 |
+
- {owner: "dotnet", name: "aspnetcore", languages: ["c#"], max_files: 2000}
|
| 173 |
+
- {owner: "dotnet", name: "runtime", languages: ["c#"], max_files: 2000}
|
| 174 |
+
- {owner: "dotnet", name: "efcore", languages: ["c#"]}
|
| 175 |
+
- {owner: "dotnet", name: "roslyn", languages: ["c#"], max_files: 2000}
|
| 176 |
+
|
| 177 |
+
# ---------- Ruby (3) ----------
|
| 178 |
+
- {owner: "rails", name: "rails", languages: ["ruby"], max_files: 2000}
|
| 179 |
+
- {owner: "ruby", name: "ruby", languages: ["c", "ruby"], max_files: 2000}
|
| 180 |
+
- {owner: "sinatra", name: "sinatra", languages: ["ruby"]}
|
| 181 |
+
|
| 182 |
+
# ---------- PHP (3) ----------
|
| 183 |
+
- {owner: "laravel", name: "framework", languages: ["php"], max_files: 2000}
|
| 184 |
+
- {owner: "symfony", name: "symfony", languages: ["php"], max_files: 2000}
|
| 185 |
+
- {owner: "php", name: "php-src", languages: ["c"], max_files: 2000}
|
| 186 |
+
|
| 187 |
+
# ---------- Swift (2) ----------
|
| 188 |
+
- {owner: "apple", name: "swift", languages: ["swift"], max_files: 2000}
|
| 189 |
+
- {owner: "vapor", name: "vapor", languages: ["swift"]}
|
| 190 |
+
|
| 191 |
+
# ---------- Kotlin (3) ----------
|
| 192 |
+
- {owner: "JetBrains", name: "kotlin", languages: ["kotlin"], max_files: 2000}
|
| 193 |
+
- {owner: "Kotlin", name: "ktor", languages: ["kotlin"]}
|
| 194 |
+
- {owner: "android", name: "architecture-components-samples", languages: ["kotlin"]}
|
| 195 |
+
|
| 196 |
+
# ---------- ML / DL / LLM (extra, 8) ----------
|
| 197 |
+
- {owner: "stanfordnlp", name: "stanford-alpaca", languages: ["python"]}
|
| 198 |
+
- {owner: "tatsu-lab", name: "stanford_alpaca", languages: ["python"]}
|
| 199 |
+
- {owner: "lm-sys", name: "FastChat", languages: ["python"]}
|
| 200 |
+
- {owner: "OpenAccess-AI-Collective", name: "axolotl", languages: ["python"]}
|
| 201 |
+
- {owner: "Lightning-AI", name: "litgpt", languages: ["python"]}
|
| 202 |
+
- {owner: "hiyouga", name: "LLaMA-Factory", languages: ["python"]}
|
| 203 |
+
- {owner: "vllm-project", name: "vllm", languages: ["python", "cpp"], max_files: 2000}
|
| 204 |
+
- {owner: "sgl-project", name: "sglang", languages: ["python", "cpp"]}
|
| 205 |
+
|
| 206 |
+
# ---------- AI agents (5) ----------
|
| 207 |
+
- {owner: "OpenHands", name: "OpenHands", languages: ["python"]}
|
| 208 |
+
- {owner: "langchain-ai", name: "langgraph", languages: ["python"]}
|
| 209 |
+
- {owner: "crewAIInc", name: "crewAI", languages: ["python"]}
|
| 210 |
+
- {owner: "microsoft", name: "autogen", languages: ["python"]}
|
| 211 |
+
- {owner: "openai", name: "openai-python", languages: ["python"]}
|
| 212 |
+
|
| 213 |
+
# ---------- DevOps / Infrastructure (8) ----------
|
| 214 |
+
- {owner: "docker", name: "docker-ce", languages: ["go"], max_files: 2000}
|
| 215 |
+
- {owner: "containerd", name: "containerd", languages: ["go"]}
|
| 216 |
+
- {owner: "opencontainers", name: "image-spec", languages: ["go"]}
|
| 217 |
+
- {owner: "cncf", name: "landscape", languages: ["yaml"]}
|
| 218 |
+
- {owner: "helm", name: "helm", languages: ["go"]}
|
| 219 |
+
- {owner: "istio", name: "istio", languages: ["go"], max_files: 2000}
|
| 220 |
+
- {owner: "envoyproxy", name: "envoy", languages: ["cpp"], max_files: 2000}
|
| 221 |
+
- {owner: "traefik", name: "traefik", languages: ["go"]}
|
| 222 |
+
|
| 223 |
+
# ---------- Database / Storage (5) ----------
|
| 224 |
+
- {owner: "postgres", name: "postgres", languages: ["c"], max_files: 2000}
|
| 225 |
+
- {owner: "mysql", name: "mysql-server", languages: ["cpp"], max_files: 2000}
|
| 226 |
+
- {owner: "sqlite", name: "sqlite", languages: ["c"]}
|
| 227 |
+
- {owner: "redis", name: "redis", languages: ["c"]}
|
| 228 |
+
- {owner: "mongodb", name: "mongo", languages: ["cpp"], max_files: 2000}
|
| 229 |
+
|
| 230 |
+
# ---------- Big data (5) ----------
|
| 231 |
+
- {owner: "apache", name: "spark", languages: ["scala"], max_files: 2000}
|
| 232 |
+
- {owner: "apache", name: "flink", languages: ["java"], max_files: 2000}
|
| 233 |
+
- {owner: "apache", name: "beam", languages: ["java", "python"]}
|
| 234 |
+
- {owner: "apache", name: "airflow", languages: ["python"], max_files: 2000}
|
| 235 |
+
- {owner: "airbnb", name: "airflow", languages: ["python"]}
|
| 236 |
+
|
| 237 |
+
# ---------- Data engineering (3) ----------
|
| 238 |
+
- {owner: "dbt-labs", name: "dbt-core", languages: ["python"]}
|
| 239 |
+
- {owner: "pallets", name: "jinja", languages: ["python"]}
|
| 240 |
+
- {owner: "great-expectations", name: "great_expectations", languages: ["python"]}
|
| 241 |
+
|
| 242 |
+
# ---------- Algorithms / data structures (5) ----------
|
| 243 |
+
- {owner: "TheAlgorithms", name: "Python", languages: ["python"], max_files: 2000}
|
| 244 |
+
- {owner: "TheAlgorithms", name: "C", languages: ["c"]}
|
| 245 |
+
- {owner: "TheAlgorithms", name: "Java", languages: ["java"]}
|
| 246 |
+
- {owner: "TheAlgorithms", name: "Go", languages: ["go"]}
|
| 247 |
+
- {owner: "keon", name: "algorithms", languages: ["python"]}
|
| 248 |
+
|
| 249 |
+
# ---------- Compilers / Languages (3) ----------
|
| 250 |
+
- {owner: "rust-lang", name: "chalk", languages: ["rust"]}
|
| 251 |
+
- {owner: "tree-sitter", name: "tree-sitter", languages: ["c", "rust"]}
|
| 252 |
+
- {owner: "vlang", name: "v", languages: ["v"]}
|
| 253 |
+
|
| 254 |
+
# ---------- Editors / IDEs (3) ----------
|
| 255 |
+
- {owner: "microsoft", name: "vscode", languages: ["typescript"], max_files: 3000}
|
| 256 |
+
- {owner: "neovim", name: "neovim", languages: ["c", "lua"], max_files: 2000}
|
| 257 |
+
- {owner: "emacs", name: "emacs", languages: ["c", "emacs-lisp"], max_files: 2000}
|
| 258 |
+
|
| 259 |
+
# ---------- DevTools (5) ----------
|
| 260 |
+
- {owner: "cli", name: "cli", languages: ["go"]}
|
| 261 |
+
- {owner: "junegunn", name: "fzf", languages: ["go"]}
|
| 262 |
+
- {owner: "tmux", name: "tmux", languages: ["c"]}
|
| 263 |
+
- {owner: "nvie", name: "gitflow", languages: ["shell"]}
|
| 264 |
+
- {owner: "nvbn", name: "thefuck", languages: ["python"]}
|
| 265 |
+
|
| 266 |
+
# ---------- Security / Crypto (3) ----------
|
| 267 |
+
- {owner: "pyca", name: "pyopenssl", languages: ["python"]}
|
| 268 |
+
- {owner: "openssl", name: "openssl", languages: ["c"], max_files: 2000}
|
| 269 |
+
- {owner: "libressl-portable", name: "openbsd", languages: ["c"]}
|
| 270 |
+
|
| 271 |
+
# ---------- Blockchain / Web3 (5) ----------
|
| 272 |
+
- {owner: "ethereum", name: "go-ethereum", languages: ["go"], max_files: 2000}
|
| 273 |
+
- {owner: "bitcoin", name: "bitcoin", languages: ["cpp"], max_files: 2000}
|
| 274 |
+
- {owner: "solana-labs", name: "solana", languages: ["rust"], max_files: 2000}
|
| 275 |
+
- {owner: "OpenZeppelin", name: "openzeppelin-contracts", languages: ["solidity"]}
|
| 276 |
+
- {owner: "chainlink", name: "contracts", languages: ["solidity"]}
|
| 277 |
+
|
| 278 |
+
# ---------- Vietnamese-specific (5) ----------
|
| 279 |
+
- {owner: "Vietnamese-data-science", name: "vdsc", languages: ["python"]}
|
| 280 |
+
- {owner: "vinbigdata-medical", name: "vinbigdata", languages: ["python"]}
|
| 281 |
+
- {owner: "undertheseanlp", name: "underthesea", languages: ["python"]}
|
| 282 |
+
- {owner: "vietai", name: "vietai-website", languages: ["python"]}
|
| 283 |
+
- {owner: "vncorenlp", name: "VnCoreNLP", languages: ["java"]}
|
| 284 |
+
|
| 285 |
+
# ---------- Open source sample projects (10) ----------
|
| 286 |
+
- {owner: "httpie", name: "httpie", languages: ["python"]}
|
| 287 |
+
- {owner: "ansible", name: "awx", languages: ["python"]}
|
| 288 |
+
- {owner: "zulip", name: "zulip", languages: ["python"], max_files: 2000}
|
| 289 |
+
- {owner: "mailpile", name: "Mailpile", languages: ["python"]}
|
| 290 |
+
- {owner: "satwikkansal", name: "wtfpython", languages: ["python"]}
|
| 291 |
+
- {owner: "karpathy", name: "nanoGPT", languages: ["python"]}
|
| 292 |
+
- {owner: "karpathy", name: "micrograd", languages: ["python"]}
|
| 293 |
+
- {owner: "milesmcc", name: "shamir-secret-sharing", languages: ["python"]}
|
| 294 |
+
- {owner: "madewithml", name: "basics", languages: ["python"]}
|
| 295 |
+
- {owner: "GokuAI", name: "alpaca-lora", languages: ["python"]}
|
| 296 |
+
|
| 297 |
+
# =============================================================================
|
| 298 |
+
# HuggingFace datasets — curated (~50)
|
| 299 |
+
# =============================================================================
|
| 300 |
+
huggingface:
|
| 301 |
+
enabled: true
|
| 302 |
+
cache_dir: "./data_cache/hf"
|
| 303 |
+
|
| 304 |
+
datasets:
|
| 305 |
+
# ---------- Code datasets (15) ----------
|
| 306 |
+
- {name: "codeparrot/codeparrot-clean", max_samples: 100000, language: "python"}
|
| 307 |
+
- {name: "codeparrot/github-code", max_samples: 50000, language: "multiple"}
|
| 308 |
+
- {name: "bigcode/the-stack-dedup", max_samples: 50000, language: "multiple"}
|
| 309 |
+
- {name: "bigcode/the-stack-v2-train-full-ids", max_samples: 20000}
|
| 310 |
+
- {name: "bigcode/starcoder2data", max_samples: 30000}
|
| 311 |
+
- {name: "nampdn-ai/tiny-codes", max_samples: 50000, language: "multiple"}
|
| 312 |
+
- {name: "HuggingFaceH4/CodeAlpaca_20K", max_samples: 20000, language: "python"}
|
| 313 |
+
- {name: "sahil2801/codealpaca", max_samples: 10000}
|
| 314 |
+
- {name: "nickroany/Evol-Instruct-Code", max_samples: 10000}
|
| 315 |
+
- {name: "iamtarun/codecontest", max_samples: 5000}
|
| 316 |
+
- {name: "openai/human-eval", max_samples: 1000}
|
| 317 |
+
- {name: "google-research-datasets/mbpp", max_samples: 1000}
|
| 318 |
+
- {name: "KaravanG/bqc-leaderboard", max_samples: 5000}
|
| 319 |
+
- {name: "bigcode/commitpackft", max_samples: 10000}
|
| 320 |
+
- {name: "bigcode/self-oss-instruct", max_samples: 10000}
|
| 321 |
+
|
| 322 |
+
# ---------- General text / web (15) ----------
|
| 323 |
+
- {name: "wikimedia/wikipedia", subset: "20231101.vi", max_samples: 50000}
|
| 324 |
+
- {name: "wikimedia/wikipedia", subset: "20231101.en", max_samples: 50000}
|
| 325 |
+
- {name: "oscar-corpus/OSCAR-2301", subset: "vi", max_samples: 30000}
|
| 326 |
+
- {name: "oscar-corpus/OSCAR-2301", subset: "en", max_samples: 30000}
|
| 327 |
+
- {name: "c4", subset: "en", max_samples: 50000}
|
| 328 |
+
- {name: "c4", subset: "vi", max_samples: 20000}
|
| 329 |
+
- {name: "allenai/dolma", max_samples: 50000}
|
| 330 |
+
- {name: "EleutherAI/pile", max_samples: 30000}
|
| 331 |
+
- {name: "HuggingFaceFW/fineweb", subset: "sample-10BT", max_samples: 50000}
|
| 332 |
+
- {name: "HuggingFaceFW/fineweb-edu", max_samples: 30000}
|
| 333 |
+
- {name: "allenai/peS2o", max_samples: 20000}
|
| 334 |
+
- {name: "allenai/dolma", subset: "v1_5-sample", max_samples: 20000}
|
| 335 |
+
- {name: "togethercomputer/RedPajama-Data-1T-Sample", max_samples: 20000}
|
| 336 |
+
- {name: "open-web-math/open-web-math", max_samples: 30000}
|
| 337 |
+
- {name: "math-ai/stack-math", max_samples: 20000}
|
| 338 |
+
|
| 339 |
+
# ---------- Instruction-tuning (15) ----------
|
| 340 |
+
- {name: "HuggingFaceH4/ultrachat_200k", max_samples: 50000}
|
| 341 |
+
- {name: "Open-Orca/OpenOrca", max_samples: 30000}
|
| 342 |
+
- {name: "teknium/OpenHermes-2.5", max_samples: 50000}
|
| 343 |
+
- {name: "databricks/databricks-dolly-15k", max_samples: 15000}
|
| 344 |
+
- {name: "tatsu-lab/alpaca", max_samples: 50000}
|
| 345 |
+
- {name: "vicgalle/configurable-system-prompts", max_samples: 10000}
|
| 346 |
+
- {name: "WizardLMTeam/WizardLM_evol_instruct_70k", max_samples: 30000}
|
| 347 |
+
- {name: "allenai/tulu-3-sft-mixture", max_samples: 50000}
|
| 348 |
+
- {name: "allenai/tulu-3-sft-personas-instruction-following", max_samples: 20000}
|
| 349 |
+
- {name: "allenai/RLVR-IFeval", max_samples: 10000}
|
| 350 |
+
- {name: "HuggingFaceH4/no_robots", max_samples: 10000}
|
| 351 |
+
- {name: "lmsys/lmsys-chat-1m", max_samples: 30000}
|
| 352 |
+
- {name: "sharegpt/sharegpt_vicuna_unfiltered", max_samples: 20000}
|
| 353 |
+
- {name: "openchat/openchat_3.5", max_samples: 10000}
|
| 354 |
+
- {name: "OpenAssistant/oasst1", max_samples: 30000}
|
| 355 |
+
|
| 356 |
+
# ---------- Math (10) ----------
|
| 357 |
+
- {name: "meta-math/MetaMathQA", max_samples: 50000}
|
| 358 |
+
- {name: "gsm8k", max_samples: 10000}
|
| 359 |
+
- {name: "lighteval/MATH", max_samples: 10000}
|
| 360 |
+
- {name: "hendrycks/competition_math", max_samples: 10000}
|
| 361 |
+
- {name: "math-ai/AQuA", max_samples: 5000}
|
| 362 |
+
- {name: "hendrycks/MATH", max_samples: 10000}
|
| 363 |
+
- {name: "openai/grade_school_math", max_samples: 8000}
|
| 364 |
+
- {name: "tasksource/strategyqa", max_samples: 5000}
|
| 365 |
+
- {name: "allenai/ai2_arc", max_samples: 5000}
|
| 366 |
+
- {name: "openai/openai_humaneval", max_samples: 1000}
|
| 367 |
+
|
| 368 |
+
# ---------- Vietnamese-specific (10) ----------
|
| 369 |
+
- {name: "vietgpt/news_corpus", max_samples: 30000}
|
| 370 |
+
- {name: "vietgpt/vietgpt-wiki", max_samples: 20000}
|
| 371 |
+
- {name: "PhoAT/PhoBERT", max_samples: 10000}
|
| 372 |
+
- {name: "vinbigdata/uit-viic", max_samples: 5000}
|
| 373 |
+
- {name: "sonlam/ Vietnamese-translation-alpaca", max_samples: 10000}
|
| 374 |
+
- {name: "nhoxquyxoem/vi-alpaca-vicuna-instruct", max_samples: 5000}
|
| 375 |
+
- {name: "VietnamAIHub/Vietnamese_translation", max_samples: 10000}
|
| 376 |
+
- {name: "vietnamese-data-science/vi-news", max_samples: 10000}
|
| 377 |
+
- {name: "duongkstn/mt-vi-train", max_samples: 5000}
|
| 378 |
+
- {name: "botran/vagrant-vi", max_samples: 5000}
|
| 379 |
+
|
| 380 |
+
# =============================================================================
|
| 381 |
+
# arXiv — scientific papers
|
| 382 |
+
# =============================================================================
|
| 383 |
+
arxiv:
|
| 384 |
+
enabled: true
|
| 385 |
+
delay_seconds: 3.0
|
| 386 |
+
|
| 387 |
+
queries:
|
| 388 |
+
- "transformer architecture"
|
| 389 |
+
- "mixture of experts"
|
| 390 |
+
- "large language model"
|
| 391 |
+
- "attention mechanism"
|
| 392 |
+
- "code generation"
|
| 393 |
+
- "program synthesis"
|
| 394 |
+
- "neural machine translation"
|
| 395 |
+
- "retrieval augmented generation"
|
| 396 |
+
- "instruction tuning"
|
| 397 |
+
- "reinforcement learning human feedback"
|
| 398 |
+
- "chain of thought reasoning"
|
| 399 |
+
- "prompt engineering"
|
| 400 |
+
- "fine-tuning language model"
|
| 401 |
+
- "quantization neural network"
|
| 402 |
+
- "knowledge distillation"
|
| 403 |
+
- "multi-agent systems"
|
| 404 |
+
- "tool use language model"
|
| 405 |
+
- "code completion"
|
| 406 |
+
- "static analysis"
|
| 407 |
+
- "program verification"
|
| 408 |
+
- "diffusion models"
|
| 409 |
+
- "vision transformer"
|
| 410 |
+
- "multimodal learning"
|
| 411 |
+
- "federated learning"
|
| 412 |
+
- "differential privacy"
|
| 413 |
+
- "graph neural network"
|
| 414 |
+
- "reinforcement learning"
|
| 415 |
+
- "meta learning"
|
| 416 |
+
- "few-shot learning"
|
| 417 |
+
- "self-supervised learning"
|
| 418 |
+
- "contrastive learning"
|
| 419 |
+
- "long context language model"
|
| 420 |
+
- "RoPE"
|
| 421 |
+
- "flash attention"
|
| 422 |
+
- "sliding window attention"
|
| 423 |
+
- "ALiBi"
|
| 424 |
+
- "RLHF"
|
| 425 |
+
- "DPO"
|
| 426 |
+
- "GRPO"
|
| 427 |
+
- "RLAIF"
|
| 428 |
+
- "agent benchmark"
|
| 429 |
+
|
| 430 |
+
# =============================================================================
|
| 431 |
+
# Wikipedia — encyclopedic text
|
| 432 |
+
# =============================================================================
|
| 433 |
+
wikipedia:
|
| 434 |
+
enabled: true
|
| 435 |
+
languages: ["vi", "en"]
|
| 436 |
+
topics:
|
| 437 |
+
vi:
|
| 438 |
+
- "Trí tuệ nhân tạo"
|
| 439 |
+
- "Học máy"
|
| 440 |
+
- "Mạng nơ-ron nhân tạo"
|
| 441 |
+
- "Python (ngôn ngữ lập trình)"
|
| 442 |
+
- "JavaScript"
|
| 443 |
+
- "Linux"
|
| 444 |
+
- "Cơ sở dữ liệu"
|
| 445 |
+
- "Thuật toán"
|
| 446 |
+
- "Cấu trúc dữ liệu"
|
| 447 |
+
- "Lập trình hướng đối tượng"
|
| 448 |
+
- "API"
|
| 449 |
+
- "JSON"
|
| 450 |
+
- "Git"
|
| 451 |
+
- "Hệ điều hành"
|
| 452 |
+
- "Học sâu"
|
| 453 |
+
- "Xử lý ngôn ngữ tự nhiên"
|
| 454 |
+
- "Big data"
|
| 455 |
+
- "Điện toán đám mây"
|
| 456 |
+
- "Cryptography"
|
| 457 |
+
- "Blockchain"
|
| 458 |
+
- "Microservices"
|
| 459 |
+
- "Docker (phần mềm)"
|
| 460 |
+
- "Kubernetes"
|
| 461 |
+
- "Terraform (phần mềm)"
|
| 462 |
+
- "Ansible"
|
| 463 |
+
- "PostgreSQL"
|
| 464 |
+
- "Redis"
|
| 465 |
+
- "MongoDB"
|
| 466 |
+
en:
|
| 467 |
+
- "Artificial intelligence"
|
| 468 |
+
- "Machine learning"
|
| 469 |
+
- "Neural network"
|
| 470 |
+
- "Python (programming language)"
|
| 471 |
+
- "JavaScript"
|
| 472 |
+
- "Linux"
|
| 473 |
+
- "Database"
|
| 474 |
+
- "Algorithm"
|
| 475 |
+
- "Data structure"
|
| 476 |
+
- "Object-oriented programming"
|
| 477 |
+
- "API"
|
| 478 |
+
- "JSON"
|
| 479 |
+
- "Git"
|
| 480 |
+
- "Operating system"
|
| 481 |
+
- "Deep learning"
|
| 482 |
+
- "Natural language processing"
|
| 483 |
+
- "Big data"
|
| 484 |
+
- "Cloud computing"
|
| 485 |
+
- "Transformer (deep learning model)"
|
| 486 |
+
- "Large language model"
|
| 487 |
+
- "Diffusion model"
|
| 488 |
+
- "GPT"
|
| 489 |
+
- "BERT"
|
| 490 |
+
- "Mixture of experts"
|
| 491 |
+
- "FlashAttention"
|
| 492 |
+
- "RoPE"
|
| 493 |
+
- "Long context language model"
|
| 494 |
+
|
| 495 |
+
# =============================================================================
|
| 496 |
+
# StackOverflow — Q&A
|
| 497 |
+
# =============================================================================
|
| 498 |
+
stackoverflow:
|
| 499 |
+
enabled: true
|
| 500 |
+
page_size: 100
|
| 501 |
+
min_score: 5
|
| 502 |
+
tags:
|
| 503 |
+
- "python"
|
| 504 |
+
- "javascript"
|
| 505 |
+
- "java"
|
| 506 |
+
- "c#"
|
| 507 |
+
- "php"
|
| 508 |
+
- "android"
|
| 509 |
+
- "html"
|
| 510 |
+
- "jquery"
|
| 511 |
+
- "c++"
|
| 512 |
+
- "css"
|
| 513 |
+
- "ios"
|
| 514 |
+
- "mysql"
|
| 515 |
+
- "sql"
|
| 516 |
+
- "node.js"
|
| 517 |
+
- "reactjs"
|
| 518 |
+
- "ruby-on-rails"
|
| 519 |
+
- "vue.js"
|
| 520 |
+
- "typescript"
|
| 521 |
+
- "docker"
|
| 522 |
+
- "git"
|
| 523 |
+
- "go"
|
| 524 |
+
- "rust"
|
| 525 |
+
- "machine-learning"
|
| 526 |
+
- "deep-learning"
|
| 527 |
+
- "pytorch"
|
| 528 |
+
- "tensorflow"
|
| 529 |
+
- "pandas"
|
| 530 |
+
- "numpy"
|
| 531 |
+
- "regex"
|
| 532 |
+
- "algorithm"
|
| 533 |
+
- "bash"
|
| 534 |
+
- "shell"
|
| 535 |
+
- "linux"
|
| 536 |
+
- "kubernetes"
|
| 537 |
+
- "terraform"
|
| 538 |
+
- "ansible"
|
| 539 |
+
- "aws"
|
| 540 |
+
- "azure"
|
| 541 |
+
- "gcp"
|
| 542 |
+
- "redis"
|
| 543 |
+
- "elasticsearch"
|
| 544 |
+
- "kafka"
|
| 545 |
+
- "rabbitmq"
|
| 546 |
+
- "postgresql"
|
| 547 |
+
- "mongodb"
|
| 548 |
+
- "sqlite"
|
| 549 |
+
|
| 550 |
+
# =============================================================================
|
| 551 |
+
# v0.3 NEW SOURCES
|
| 552 |
+
# =============================================================================
|
| 553 |
+
|
| 554 |
+
# The-Stack v2 — BigCode's massive code dataset
|
| 555 |
+
the_stack:
|
| 556 |
+
enabled: true
|
| 557 |
+
cache_dir: "./data_cache/the_stack"
|
| 558 |
+
version: "v2"
|
| 559 |
+
# Top languages by sample count (rest skipped to keep size manageable)
|
| 560 |
+
languages:
|
| 561 |
+
- "python"
|
| 562 |
+
- "javascript"
|
| 563 |
+
- "typescript"
|
| 564 |
+
- "java"
|
| 565 |
+
- "go"
|
| 566 |
+
- "rust"
|
| 567 |
+
- "c"
|
| 568 |
+
- "cpp"
|
| 569 |
+
- "csharp"
|
| 570 |
+
- "ruby"
|
| 571 |
+
- "php"
|
| 572 |
+
- "swift"
|
| 573 |
+
- "kotlin"
|
| 574 |
+
- "scala"
|
| 575 |
+
- "shell"
|
| 576 |
+
- "sql"
|
| 577 |
+
max_samples_per_language: 5000
|
| 578 |
+
min_stars: 0 # include all repos regardless of stars
|
| 579 |
+
license_filter: ["mit", "apache-2.0", "bsd-3-clause", "bsd-2-clause", "mpl-2.0", "unlicense"]
|
| 580 |
+
|
| 581 |
+
# StarCoder2 training data (github-code + commits + notebooks)
|
| 582 |
+
starcoder2_data:
|
| 583 |
+
enabled: true
|
| 584 |
+
cache_dir: "./data_cache/starcoder2"
|
| 585 |
+
components:
|
| 586 |
+
- "github_code" # code files
|
| 587 |
+
- "github_commits" # commit diffs (good for editing tasks)
|
| 588 |
+
- "github_jupyter" # notebook cells (markdown + code)
|
| 589 |
+
max_samples_per_component: 20000
|
| 590 |
+
languages:
|
| 591 |
+
- "python"
|
| 592 |
+
- "javascript"
|
| 593 |
+
- "typescript"
|
| 594 |
+
- "java"
|
| 595 |
+
- "go"
|
| 596 |
+
- "rust"
|
| 597 |
+
- "c"
|
| 598 |
+
- "cpp"
|
| 599 |
+
|
| 600 |
+
# Python-Alpaca — high-quality Python instruction data
|
| 601 |
+
python_alpaca:
|
| 602 |
+
enabled: true
|
| 603 |
+
cache_dir: "./data_cache/python_alpaca"
|
| 604 |
+
sources:
|
| 605 |
+
- {name: "sahil2801/codealpaca", max_samples: 20000}
|
| 606 |
+
- {name: "HuggingFaceH4/CodeAlpaca_20K", max_samples: 20000}
|
| 607 |
+
- {name: "nickroany/Evol-Instruct-Code", max_samples: 15000}
|
| 608 |
+
- {name: "TheBloke/CodeAlpaca-13B", max_samples: 5000}
|
| 609 |
+
- {name: "codeparrot/codeparrot-clean", max_samples: 50000}
|
| 610 |
+
- {name: "nampdn-ai/tiny-codes", max_samples: 50000}
|
| 611 |
+
|
| 612 |
+
# Kaggle — competition kernels & datasets metadata
|
| 613 |
+
kaggle:
|
| 614 |
+
enabled: false # disabled by default — requires API key
|
| 615 |
+
cache_dir: "./data_cache/kaggle"
|
| 616 |
+
api_key_env: "KAGGLE_API_KEY"
|
| 617 |
+
competitions:
|
| 618 |
+
- "titanic"
|
| 619 |
+
- "house-prices-advanced-regression-techniques"
|
| 620 |
+
- "digit-recognizer"
|
| 621 |
+
- " Spaceship-Titanic"
|
| 622 |
+
- "favorita-grocery-sales-forecasting"
|
| 623 |
+
max_kernels_per_competition: 100
|
| 624 |
+
|
| 625 |
+
# =============================================================================
|
| 626 |
+
# Processing pipeline settings
|
| 627 |
+
# =============================================================================
|
| 628 |
+
processing:
|
| 629 |
+
cleaner:
|
| 630 |
+
remove_html: true
|
| 631 |
+
remove_urls: false
|
| 632 |
+
normalize_whitespace: true
|
| 633 |
+
min_length: 50
|
| 634 |
+
max_length: 100000
|
| 635 |
+
|
| 636 |
+
quality_filter:
|
| 637 |
+
min_length: 50
|
| 638 |
+
max_length: 100000
|
| 639 |
+
min_words: 10
|
| 640 |
+
min_unique_ratio: 0.3
|
| 641 |
+
max_repetition: 0.5
|
| 642 |
+
|
| 643 |
+
deduplicator:
|
| 644 |
+
ngram_size: 5
|
| 645 |
+
num_perm: 128
|
| 646 |
+
similarity_threshold: 0.8
|
| 647 |
+
|
| 648 |
+
# v0.3 NEW processors
|
| 649 |
+
language_id:
|
| 650 |
+
enabled: true
|
| 651 |
+
# Identify language of each text sample (drops mislabelled)
|
| 652 |
+
min_confidence: 0.85
|
| 653 |
+
allowed_languages: ["vi", "en", "code"]
|
| 654 |
+
|
| 655 |
+
code_quality:
|
| 656 |
+
enabled: true
|
| 657 |
+
# Score code samples (1-10), drop samples below threshold
|
| 658 |
+
min_score: 6.0
|
| 659 |
+
factors:
|
| 660 |
+
has_docstring: 1.5
|
| 661 |
+
has_type_hints: 1.0
|
| 662 |
+
no_print: 0.5
|
| 663 |
+
no_eval: 1.0
|
| 664 |
+
no_bare_except: 1.0
|
| 665 |
+
reasonable_length: 1.0 # 10-500 lines
|
| 666 |
+
has_test: 2.0 # bonus for adjacent test file
|
| 667 |
+
|
| 668 |
+
curriculum:
|
| 669 |
+
stages:
|
| 670 |
+
- {name: "easy", min_length: 50, max_length: 500, min_quality: 0.7}
|
| 671 |
+
- {name: "medium", min_length: 500, max_length: 5000, min_quality: 0.6}
|
| 672 |
+
- {name: "hard", min_length: 5000, max_length: 30000, min_quality: 0.7}
|
| 673 |
+
- {name: "expert", min_length: 30000, max_length: 100000, min_quality: 0.8}
|
| 674 |
+
|
| 675 |
+
# =============================================================================
|
| 676 |
+
# Token budget estimation (v0.3 NEW)
|
| 677 |
+
# =============================================================================
|
| 678 |
+
token_budget:
|
| 679 |
+
total_target_tokens: 500_000_000_000 # 500B tokens (for 30B model pretrain)
|
| 680 |
+
distribution:
|
| 681 |
+
code: 0.40 # 40% code (The-Stack, StarCoder2-data, GitHub)
|
| 682 |
+
text: 0.30 # 30% natural text (Wikipedia, C4, OSCAR)
|
| 683 |
+
instruction: 0.15 # 15% instruction-tuning (Alpaca, ShareGPT)
|
| 684 |
+
math: 0.10 # 10% math (GSM8K, MATH, MetaMathQA)
|
| 685 |
+
vietnamese: 0.05 # 5% Vietnamese-specific
|
data/README.md
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Data directory
|
| 2 |
+
|
| 3 |
+
Thư mục này chứa dữ liệu huấn luyện (nếu có).
|
| 4 |
+
|
| 5 |
+
This directory contains training data (if any).
|
| 6 |
+
|
| 7 |
+
Hiện tại, training data được hardcoded trong `nexus/training/dataset.py`.
|
| 8 |
+
|
| 9 |
+
Currently, training data is hardcoded in `nexus/training/dataset.py`.
|
docs/ARCHITECTURE.md
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Kiến trúc Nexus Coder / Nexus Coder Architecture
|
| 2 |
+
|
| 3 |
+
## Tổng quan / Overview
|
| 4 |
+
|
| 5 |
+
Nexus Coder v0.1 sử dụng kiến trúc **Mixture of Experts (MoE) Transformer** tương tự Mixtral 8x7B và DeepSeek-V3.
|
| 6 |
+
|
| 7 |
+
Nexus Coder v0.1 uses a **Mixture of Experts (MoE) Transformer** architecture similar to Mixtral 8x7B and DeepSeek-V3.
|
| 8 |
+
|
| 9 |
+
## Các thành phần / Components
|
| 10 |
+
|
| 11 |
+
### 1. Token Embedding
|
| 12 |
+
- Vocab size: 32,000
|
| 13 |
+
- Hidden size: 2,048
|
| 14 |
+
- Tokens được nhúng thành vector 2048 chiều
|
| 15 |
+
|
| 16 |
+
### 2. Grouped Query Attention (GQA)
|
| 17 |
+
- 16 query heads
|
| 18 |
+
- 4 KV heads (ratio 4:1)
|
| 19 |
+
- Head dimension: 128
|
| 20 |
+
- Giảm 4x memory cho KV cache so với MHA truyền thống
|
| 21 |
+
|
| 22 |
+
### 3. Rotary Position Embedding (RoPE)
|
| 23 |
+
- Base: 10,000
|
| 24 |
+
- Hỗ trợ tối đa 50,000 positions
|
| 25 |
+
- Cho phép model hiểu vị trí tương đối giữa các tokens
|
| 26 |
+
|
| 27 |
+
### 4. RMSNorm
|
| 28 |
+
- Thay thế LayerNorm truyền thống
|
| 29 |
+
- Không có bias, không trừ mean
|
| 30 |
+
- Nhanh hơn ~10-20%
|
| 31 |
+
|
| 32 |
+
### 5. SwiGLU Activation
|
| 33 |
+
- `SiLU(gate(x)) * up(x)`
|
| 34 |
+
- Hiệu quả hơn ReLU/GELU
|
| 35 |
+
- Có 3 ma trận: gate, up, down (3 * hidden * intermediate params)
|
| 36 |
+
|
| 37 |
+
### 6. Mixture of Experts (MoE) - Cốt lõi
|
| 38 |
+
- **24 experts** tổng cộng (mỗi expert là một SwiGLU FFN)
|
| 39 |
+
- **3 active experts** mỗi token (top-3 routing)
|
| 40 |
+
- Router: linear layer (hidden_size → num_experts)
|
| 41 |
+
- Load balancing loss: auxiliary loss để tránh expert collapse
|
| 42 |
+
|
| 43 |
+
#### Routing Algorithm
|
| 44 |
+
```
|
| 45 |
+
1. Router tính gate_logits = W_router @ x
|
| 46 |
+
2. routing_weights = softmax(gate_logits)
|
| 47 |
+
3. top_k_weights, top_k_indices = topk(routing_weights, k=3)
|
| 48 |
+
4. Normalize top_k_weights
|
| 49 |
+
5. Mỗi token đi qua 3 expert được chọn
|
| 50 |
+
6. Output = sum(weight_i * expert_i(x))
|
| 51 |
+
```
|
| 52 |
+
|
| 53 |
+
## Tính toán tham số / Parameter Math
|
| 54 |
+
|
| 55 |
+
```
|
| 56 |
+
Embedding: vocab_size × hidden = 32000 × 2048 = 65.5M
|
| 57 |
+
Per layer attn: 2048² + 2×(2048×512) + 2048² = 10.5M (Q, K, V, O with GQA)
|
| 58 |
+
Per expert: 3 × 2048 × 5632 = 34.6M (gate + up + down)
|
| 59 |
+
Per layer MoE: 24 × 34.6M = 830M (total)
|
| 60 |
+
3 × 34.6M = 104M (active)
|
| 61 |
+
Per layer total: 10.5M + 830M = 840.5M
|
| 62 |
+
12 layers: 10,086M
|
| 63 |
+
LM head: 65.5M
|
| 64 |
+
────────────────────────────────────
|
| 65 |
+
TOTAL: 10,223M ≈ 10.22B ✓
|
| 66 |
+
ACTIVE: 65.5 + 12×(10.5 + 104) + 65.5 = 1,503M ≈ 1.50B ✓
|
| 67 |
+
```
|
| 68 |
+
|
| 69 |
+
## Workflow
|
| 70 |
+
|
| 71 |
+
### Training Workflow
|
| 72 |
+
1. Tokenize input text → token IDs
|
| 73 |
+
2. Embed tokens → hidden states [B, L, H]
|
| 74 |
+
3. For each layer:
|
| 75 |
+
- Pre-norm → Attention → residual
|
| 76 |
+
- Pre-norm → MoE (router + experts) → residual
|
| 77 |
+
4. Final norm → LM head → logits
|
| 78 |
+
5. Compute cross-entropy loss + aux loss
|
| 79 |
+
6. Backpropagation
|
| 80 |
+
|
| 81 |
+
### Inference Workflow
|
| 82 |
+
1. Tokenize prompt
|
| 83 |
+
2. Forward pass through all layers
|
| 84 |
+
3. Get logits for last position
|
| 85 |
+
4. Apply temperature, top-k, top-p
|
| 86 |
+
5. Sample next token
|
| 87 |
+
6. Append to sequence, repeat
|
| 88 |
+
|
| 89 |
+
## Tối ưu / Optimizations
|
| 90 |
+
|
| 91 |
+
- **KV Cache**: Cache K, V từ các token trước để tăng tốc generation
|
| 92 |
+
- **GQA**: Giảm memory và computation cho attention
|
| 93 |
+
- **Pre-norm**: Ổn định hơn post-norm trong training
|
| 94 |
+
- **Mixed Precision**: Hỗ trợ fp16/bf16 để tiết kiệm memory
|
| 95 |
+
- **Gradient Checkpointing**: Đánh đổi compute lấy memory (chưa implement trong v0.1)
|
docs/DATA.md
ADDED
|
@@ -0,0 +1,188 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Data Pipeline Documentation
|
| 2 |
+
|
| 3 |
+
Nexus Coder v0.2 có pipeline thu thập và xử lý training data hoàn chỉnh.
|
| 4 |
+
|
| 5 |
+
## Overview
|
| 6 |
+
|
| 7 |
+
```
|
| 8 |
+
┌─────────────┐ ┌──────────────┐ ┌─────────────┐ ┌──────────────┐
|
| 9 |
+
│ COLLECT │ ──> │ PROCESS │ ──> │ TRAIN │ ──> │ EVALUATE │
|
| 10 |
+
│ (5 sources) │ │ (4 stages) │ │ (curriculum)│ │ (8 benches) │
|
| 11 |
+
└─────────────┘ └──────────────┘ └─────────────┘ └──────────────┘
|
| 12 |
+
```
|
| 13 |
+
|
| 14 |
+
## Sources (Collectors)
|
| 15 |
+
|
| 16 |
+
### 1. GitHub
|
| 17 |
+
- **60+ curated repos** (Python, JS, TS, Go, Rust, C, C++)
|
| 18 |
+
- Categories: Python core, Data science, ML/DL, Web, CLI, Async, Database, Tools
|
| 19 |
+
- Quality filter: size, content, auto-generated detection
|
| 20 |
+
- File extensions: .py, .js, .ts, .go, .rs, .java, .c, .cpp, .sql, .sh, .md
|
| 21 |
+
|
| 22 |
+
### 2. HuggingFace
|
| 23 |
+
- **20+ curated datasets**:
|
| 24 |
+
- Code: codeparrot, the-stack, CodeAlpaca
|
| 25 |
+
- Text: Wikipedia (vi, en), C4, OSCAR
|
| 26 |
+
- Chat: UltraChat, OpenOrca, OpenHermes, Dolly
|
| 27 |
+
- Math: MetaMathQA, GSM8K, MATH
|
| 28 |
+
- Vietnamese: news_corpus, PhoATC
|
| 29 |
+
|
| 30 |
+
### 3. arXiv
|
| 31 |
+
- 20 curated queries (transformer, MoE, LLM, code generation, etc.)
|
| 32 |
+
- Categories: cs.CL, cs.LG, cs.AI, cs.SE, cs.PL, cs.CV, stat.ML
|
| 33 |
+
- Rate limit: 1 request per 3 seconds
|
| 34 |
+
|
| 35 |
+
### 4. Wikipedia
|
| 36 |
+
- Vietnamese + English
|
| 37 |
+
- 20 curated topics per language
|
| 38 |
+
- Random article collection supported
|
| 39 |
+
|
| 40 |
+
### 5. StackOverflow
|
| 41 |
+
- 30 curated tags (python, javascript, java, etc.)
|
| 42 |
+
- Filter by minimum score (default: 5)
|
| 43 |
+
- Includes accepted answers
|
| 44 |
+
- Rate limit: 30 req/s
|
| 45 |
+
|
| 46 |
+
## Processing Pipeline
|
| 47 |
+
|
| 48 |
+
### Stage 1: Clean (TextCleaner)
|
| 49 |
+
- HTML tag removal
|
| 50 |
+
- Unicode normalization (NFC)
|
| 51 |
+
- Control character removal
|
| 52 |
+
- HTML entity decoding
|
| 53 |
+
- Whitespace normalization
|
| 54 |
+
- Encoding fix
|
| 55 |
+
|
| 56 |
+
### Stage 2: Format (CodeFormatter)
|
| 57 |
+
- Language detection (by extension + patterns)
|
| 58 |
+
- Trailing whitespace removal
|
| 59 |
+
- Excessive blank line removal (max 2 consecutive)
|
| 60 |
+
- Leading/trailing blank line removal
|
| 61 |
+
- Markdown fence wrapping
|
| 62 |
+
|
| 63 |
+
### Stage 3: Quality Filter (QualityFilter)
|
| 64 |
+
- Length check (50-100,000 chars)
|
| 65 |
+
- Word count (min 10)
|
| 66 |
+
- Unique word ratio (min 0.3)
|
| 67 |
+
- Repetition score (max 0.5)
|
| 68 |
+
- Spam pattern detection
|
| 69 |
+
- Code presence bonus
|
| 70 |
+
|
| 71 |
+
### Stage 4: Deduplicate (Deduplicator)
|
| 72 |
+
- Exact hash dedup (MD5)
|
| 73 |
+
- MinHash LSH for near-duplicates
|
| 74 |
+
- 128 permutations, 5-gram
|
| 75 |
+
- Jaccard threshold: 0.8
|
| 76 |
+
|
| 77 |
+
## Curriculum Learning
|
| 78 |
+
|
| 79 |
+
4-stage curriculum:
|
| 80 |
+
|
| 81 |
+
| Stage | Difficulty | Length | Quality | Description |
|
| 82 |
+
|-------|-----------|--------|---------|-------------|
|
| 83 |
+
| 1 | EASY | 50-500 | ≥0.7 | Short basic text - vocabulary |
|
| 84 |
+
| 2 | MEDIUM | 500-5000 | ≥0.6 | Standard length - grammar |
|
| 85 |
+
| 3 | HARD | 5000-30000 | ≥0.7 | Long technical - deep understanding |
|
| 86 |
+
| 4 | EXPERT | 30000-100000 | ≥0.8 | Multi-step reasoning |
|
| 87 |
+
|
| 88 |
+
## Usage
|
| 89 |
+
|
| 90 |
+
### Collect raw data
|
| 91 |
+
|
| 92 |
+
```bash
|
| 93 |
+
# Collect from all sources
|
| 94 |
+
python scripts/collect_data.py --source all --output ./data/raw
|
| 95 |
+
|
| 96 |
+
# Or specific source
|
| 97 |
+
python scripts/collect_data.py --source github --max-repos 10
|
| 98 |
+
python scripts/collect_data.py --source huggingface --max-datasets 5
|
| 99 |
+
```
|
| 100 |
+
|
| 101 |
+
### Process raw data
|
| 102 |
+
|
| 103 |
+
```bash
|
| 104 |
+
python scripts/prepare_dataset.py --input ./data/raw --output ./data/processed
|
| 105 |
+
```
|
| 106 |
+
|
| 107 |
+
### Train with external data
|
| 108 |
+
|
| 109 |
+
```bash
|
| 110 |
+
python scripts/train.py --config large --include-external --steps 5000
|
| 111 |
+
```
|
| 112 |
+
|
| 113 |
+
## Output Format
|
| 114 |
+
|
| 115 |
+
Processed data saved as JSONL files by difficulty:
|
| 116 |
+
|
| 117 |
+
```
|
| 118 |
+
data/processed/
|
| 119 |
+
├── train_easy.jsonl # Stage 1 samples
|
| 120 |
+
├── train_medium.jsonl # Stage 2 samples
|
| 121 |
+
├── train_hard.jsonl # Stage 3 samples
|
| 122 |
+
├── train_expert.jsonl # Stage 4 samples
|
| 123 |
+
└── processing_stats.json # Statistics
|
| 124 |
+
```
|
| 125 |
+
|
| 126 |
+
Each JSONL line:
|
| 127 |
+
```json
|
| 128 |
+
{
|
| 129 |
+
"text": "...",
|
| 130 |
+
"source": "github:python/cpython",
|
| 131 |
+
"language": "python",
|
| 132 |
+
"metadata": {
|
| 133 |
+
"file_path": "Lib/os.py",
|
| 134 |
+
"size": 45678,
|
| 135 |
+
"quality_score": 0.85,
|
| 136 |
+
"quality": {"score": 0.85, "length": 45678, "word_count": 1200, "has_code": true},
|
| 137 |
+
"cleaned": true,
|
| 138 |
+
"cleaned_length": 45678,
|
| 139 |
+
"formatted": true,
|
| 140 |
+
"detected_language": "python"
|
| 141 |
+
}
|
| 142 |
+
}
|
| 143 |
+
```
|
| 144 |
+
|
| 145 |
+
## Environment Variables
|
| 146 |
+
|
| 147 |
+
```bash
|
| 148 |
+
# GitHub API (for search)
|
| 149 |
+
export GITHUB_TOKEN=ghp_xxx
|
| 150 |
+
|
| 151 |
+
# HuggingFace Hub (for gated datasets)
|
| 152 |
+
export HF_TOKEN=hf_xxx
|
| 153 |
+
|
| 154 |
+
# Web search API (optional)
|
| 155 |
+
export SEARCH_API_KEY=xxx
|
| 156 |
+
export BRAVE_SEARCH_API_KEY=xxx
|
| 157 |
+
```
|
| 158 |
+
|
| 159 |
+
## Estimate Data Volume
|
| 160 |
+
|
| 161 |
+
| Source | Estimated samples | Estimated size |
|
| 162 |
+
|--------|------------------|----------------|
|
| 163 |
+
| GitHub (60 repos) | ~50,000 files | ~500 MB |
|
| 164 |
+
| HuggingFace (20 datasets) | ~200,000 samples | ~2 GB (streamed) |
|
| 165 |
+
| arXiv (20 queries) | ~400 papers | ~50 MB |
|
| 166 |
+
| Wikipedia (vi+en) | ~40 articles | ~5 MB |
|
| 167 |
+
| StackOverflow (30 tags) | ~1,500 Q&A | ~10 MB |
|
| 168 |
+
| **Total** | **~250,000 samples** | **~2.5 GB** |
|
| 169 |
+
|
| 170 |
+
After deduplication and quality filter: ~150,000 high-quality samples.
|
| 171 |
+
|
| 172 |
+
## Custom Sources
|
| 173 |
+
|
| 174 |
+
Add your own collector:
|
| 175 |
+
|
| 176 |
+
```python
|
| 177 |
+
from nexus.data.collectors.base import Collector
|
| 178 |
+
|
| 179 |
+
class MyCollector(Collector):
|
| 180 |
+
def collect(self):
|
| 181 |
+
# Yield samples as dicts
|
| 182 |
+
yield {
|
| 183 |
+
"text": "...",
|
| 184 |
+
"source": "my_source",
|
| 185 |
+
"language": "en",
|
| 186 |
+
"metadata": {...},
|
| 187 |
+
}
|
| 188 |
+
```
|
docs/SKILLS.md
ADDED
|
@@ -0,0 +1,175 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Skills Documentation
|
| 2 |
+
|
| 3 |
+
Nexus Coder v0.2 có 15 skills chuyên môn, được tổ chức theo 5 categories.
|
| 4 |
+
|
| 5 |
+
## Categories
|
| 6 |
+
|
| 7 |
+
| Category | Skills |
|
| 8 |
+
|----------|--------|
|
| 9 |
+
| CODE | code_generation, code_review, code_refactor, debugging, documentation, testing |
|
| 10 |
+
| REASONING | algorithm_design, reasoning, math_skill |
|
| 11 |
+
| LANGUAGE | translation, summarization |
|
| 12 |
+
| DATA | data_analysis, sql_generation |
|
| 13 |
+
| SECURITY | security_audit |
|
| 14 |
+
| DEVOPS | performance_optimization |
|
| 15 |
+
|
| 16 |
+
## Skill List
|
| 17 |
+
|
| 18 |
+
### 1. code_generation
|
| 19 |
+
- **Category**: CODE
|
| 20 |
+
- **Priority**: HIGH
|
| 21 |
+
- **Description**: Sinh code từ mô tả tự nhiên
|
| 22 |
+
- **Languages**: Python, JavaScript, TypeScript, Go, Rust, C++, Java, SQL
|
| 23 |
+
- **Example**: "Viết hàm Python tính fibonacci"
|
| 24 |
+
|
| 25 |
+
### 2. code_review
|
| 26 |
+
- **Category**: CODE
|
| 27 |
+
- **Priority**: HIGH
|
| 28 |
+
- **Description**: Review code toàn diện
|
| 29 |
+
- **Checks**: bugs, security, performance, style, error handling, type safety
|
| 30 |
+
- **Example**: "Review đoạn code này giúp tôi"
|
| 31 |
+
|
| 32 |
+
### 3. code_refactor
|
| 33 |
+
- **Category**: CODE
|
| 34 |
+
- **Priority**: MEDIUM
|
| 35 |
+
- **Description**: Refactor code an toàn
|
| 36 |
+
- **Patterns**: Extract Method/Class, Rename, Move, Replace Conditional, etc.
|
| 37 |
+
- **Example**: "Refactor hàm này cho clean hơn"
|
| 38 |
+
|
| 39 |
+
### 4. debugging
|
| 40 |
+
- **Category**: CODE
|
| 41 |
+
- **Priority**: CRITICAL
|
| 42 |
+
- **Description**: Debug code với 7-step protocol
|
| 43 |
+
- **Supports**: Python, JavaScript, Java, Go, Rust, C++, Ruby
|
| 44 |
+
- **Example**: "Fix lỗi IndexError trong hàm này"
|
| 45 |
+
|
| 46 |
+
### 5. documentation
|
| 47 |
+
- **Category**: CODE
|
| 48 |
+
- **Priority**: MEDIUM
|
| 49 |
+
- **Description**: Sinh tài liệu tự động
|
| 50 |
+
- **Types**: Docstrings (Google/NumPy/Sphinx), README, API ref, tutorials
|
| 51 |
+
- **Example**: "Sinh docstring cho hàm này"
|
| 52 |
+
|
| 53 |
+
### 6. testing
|
| 54 |
+
- **Category**: CODE
|
| 55 |
+
- **Priority**: HIGH
|
| 56 |
+
- **Description**: Sinh tests
|
| 57 |
+
- **Types**: unit, integration, E2E, property-based, mutation, fuzz, snapshot
|
| 58 |
+
- **Frameworks**: pytest, unittest, jest, vitest, mocha, cargo test, JUnit
|
| 59 |
+
- **Example**: "Viết unit tests cho class User"
|
| 60 |
+
|
| 61 |
+
### 7. algorithm_design
|
| 62 |
+
- **Category**: REASONING
|
| 63 |
+
- **Priority**: MEDIUM
|
| 64 |
+
- **Description**: Thiết kế thuật toán
|
| 65 |
+
- **Approaches**: Brute force, Greedy, D&C, DP, Backtracking, Graph algorithms
|
| 66 |
+
- **Example**: "Tối ưu thuật toán này từ O(n²) xuống O(n log n)"
|
| 67 |
+
|
| 68 |
+
### 8. data_analysis
|
| 69 |
+
- **Category**: DATA
|
| 70 |
+
- **Priority**: MEDIUM
|
| 71 |
+
- **Description**: Phân tích dữ liệu
|
| 72 |
+
- **Steps**: Loading, cleaning, statistics, correlation, outliers, visualization
|
| 73 |
+
- **Libraries**: pandas, numpy, scipy, matplotlib, seaborn, plotly
|
| 74 |
+
- **Example**: "Phân tích dataset này và tìm insights"
|
| 75 |
+
|
| 76 |
+
### 9. translation
|
| 77 |
+
- **Category**: LANGUAGE
|
| 78 |
+
- **Priority**: MEDIUM
|
| 79 |
+
- **Description**: Dịch song ngữ Việt-Anh
|
| 80 |
+
- **Pairs**: vi↔en, vi↔zh, vi↔ja, vi↔ko, vi↔fr
|
| 81 |
+
- **Example**: "Dịch đoạn văn này sang tiếng Anh"
|
| 82 |
+
|
| 83 |
+
### 10. summarization
|
| 84 |
+
- **Category**: LANGUAGE
|
| 85 |
+
- **Priority**: MEDIUM
|
| 86 |
+
- **Description**: Tóm tắt văn bản
|
| 87 |
+
- **Methods**: extractive, abstractive, key phrase, topic modeling
|
| 88 |
+
- **Example**: "Tóm tắt bài viết này trong 3 câu"
|
| 89 |
+
|
| 90 |
+
### 11. reasoning
|
| 91 |
+
- **Category**: REASONING
|
| 92 |
+
- **Priority**: HIGH
|
| 93 |
+
- **Description**: Suy luận đa bước
|
| 94 |
+
- **Strategies**: CoT, ToT, Self-Consistency, Reflexion, ReAct, Least-to-Most
|
| 95 |
+
- **Example**: "Tại sao bầu trời màu xanh?"
|
| 96 |
+
|
| 97 |
+
### 12. math_skill
|
| 98 |
+
- **Category**: REASONING
|
| 99 |
+
- **Priority**: HIGH
|
| 100 |
+
- **Description**: Giải toán đa cấp
|
| 101 |
+
- **Domains**: arithmetic, algebra, calculus, linear algebra, probability, statistics
|
| 102 |
+
- **Tools**: sympy, numpy, scipy
|
| 103 |
+
- **Example**: "Tính đạo hàm của x³ + 2x²"
|
| 104 |
+
|
| 105 |
+
### 13. sql_generation
|
| 106 |
+
- **Category**: DATA
|
| 107 |
+
- **Priority**: HIGH
|
| 108 |
+
- **Description**: Sinh SQL queries
|
| 109 |
+
- **Dialects**: PostgreSQL, MySQL, SQLite, SQL Server, Oracle, BigQuery, Snowflake
|
| 110 |
+
- **Example**: "Viết SQL tìm top 10 khách hàng"
|
| 111 |
+
|
| 112 |
+
### 14. security_audit
|
| 113 |
+
- **Category**: SECURITY
|
| 114 |
+
- **Priority**: CRITICAL
|
| 115 |
+
- **Description**: Audit bảo mật
|
| 116 |
+
- **Standards**: OWASP Top 10, SAST, dependency vulnerabilities
|
| 117 |
+
- **Tools**: bandit, semgrep, safety, pip-audit, trufflehog
|
| 118 |
+
- **Example**: "Audit code này cho security issues"
|
| 119 |
+
|
| 120 |
+
### 15. performance_optimization
|
| 121 |
+
- **Category**: DEVOPS
|
| 122 |
+
- **Priority**: MEDIUM
|
| 123 |
+
- **Description**: Tối ưu hiệu năng
|
| 124 |
+
- **Categories**: algorithmic, memory, concurrency, caching, I/O, Python-specific
|
| 125 |
+
- **Tools**: cProfile, line_profiler, memory_profiler, py-spy
|
| 126 |
+
- **Example**: "Tối ưu hàm này đang chạy chậm"
|
| 127 |
+
|
| 128 |
+
## Usage
|
| 129 |
+
|
| 130 |
+
```python
|
| 131 |
+
from nexus.skills import get_global_registry
|
| 132 |
+
from nexus.skills.base import SkillContext
|
| 133 |
+
|
| 134 |
+
registry = get_global_registry()
|
| 135 |
+
|
| 136 |
+
# List all skills
|
| 137 |
+
print(registry.list_skills())
|
| 138 |
+
|
| 139 |
+
# Route prompt to best skill
|
| 140 |
+
skill = registry.route("Viết hàm Python tính giai thừa")
|
| 141 |
+
print(f"Selected: {skill.name}")
|
| 142 |
+
|
| 143 |
+
# Execute skill
|
| 144 |
+
context = SkillContext(prompt="Viết hàm Python tính giai thừa")
|
| 145 |
+
result = skill.execute(context)
|
| 146 |
+
print(result.output)
|
| 147 |
+
```
|
| 148 |
+
|
| 149 |
+
## Custom Skills
|
| 150 |
+
|
| 151 |
+
Tạo skill tùy chỉnh:
|
| 152 |
+
|
| 153 |
+
```python
|
| 154 |
+
from nexus.skills.base import Skill, SkillResult, SkillContext, SkillCategory, SkillPriority
|
| 155 |
+
|
| 156 |
+
class MyCustomSkill(Skill):
|
| 157 |
+
category = SkillCategory.CODE
|
| 158 |
+
priority = SkillPriority.MEDIUM
|
| 159 |
+
keywords = ["custom", "riêng"]
|
| 160 |
+
|
| 161 |
+
@property
|
| 162 |
+
def name(self) -> str:
|
| 163 |
+
return "my_custom_skill"
|
| 164 |
+
|
| 165 |
+
@property
|
| 166 |
+
def description(self) -> str:
|
| 167 |
+
return "My custom skill description"
|
| 168 |
+
|
| 169 |
+
def execute(self, context: SkillContext) -> SkillResult:
|
| 170 |
+
return SkillResult(success=True, output="Custom result")
|
| 171 |
+
|
| 172 |
+
# Register
|
| 173 |
+
from nexus.skills import get_global_registry
|
| 174 |
+
get_global_registry().register(MyCustomSkill())
|
| 175 |
+
```
|
docs/TOOLS.md
ADDED
|
@@ -0,0 +1,162 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Tools Documentation
|
| 2 |
+
|
| 3 |
+
Nexus Coder v0.2 có 18+ tools để tương tác với môi trường.
|
| 4 |
+
|
| 5 |
+
## Safety Levels
|
| 6 |
+
|
| 7 |
+
| Level | Icon | Description |
|
| 8 |
+
|-------|------|-------------|
|
| 9 |
+
| SAFE | ✓ | Read-only, no side effects |
|
| 10 |
+
| MODERATE | ⚠ | Writes to local files |
|
| 11 |
+
| DANGEROUS | ⚡ | Executes commands, network ops |
|
| 12 |
+
| DESTRUCTIVE | 💀 | Can delete data, requires confirmation |
|
| 13 |
+
|
| 14 |
+
## Tools by Category
|
| 15 |
+
|
| 16 |
+
### FILE Operations
|
| 17 |
+
- `file_read` (✓) - Đọc file text
|
| 18 |
+
- `file_write` (⚠) - Ghi file (overwrite/append)
|
| 19 |
+
- `file_list` (✓) - Liệt kê files với glob
|
| 20 |
+
- `file_delete` (💀) - Xóa file/thư mục
|
| 21 |
+
|
| 22 |
+
### EXEC
|
| 23 |
+
- `shell_exec` (⚡) - Execute bash commands
|
| 24 |
+
- `python_exec` (⚡) - Execute Python code (sandboxed)
|
| 25 |
+
- `git_ops` (⚡) - Git commands
|
| 26 |
+
|
| 27 |
+
### WEB
|
| 28 |
+
- `http_request` (⚠) - HTTP GET/POST/PUT/DELETE
|
| 29 |
+
- `web_fetch` (✓) - Fetch webpage, extract text
|
| 30 |
+
- `web_search` (✓) - Search web
|
| 31 |
+
|
| 32 |
+
### CODE
|
| 33 |
+
- `code_search` (✓) - Regex search trong code
|
| 34 |
+
- `code_lint` (✓) - Lint code (ruff, flake8, pylint)
|
| 35 |
+
- `code_format` (⚠) - Format code (black, autopep8, isort)
|
| 36 |
+
- `regex_search` (✓) - Regex search trong files
|
| 37 |
+
|
| 38 |
+
### MATH
|
| 39 |
+
- `calculator` (✓) - Safe math expression eval
|
| 40 |
+
|
| 41 |
+
### PARSER
|
| 42 |
+
- `json_parse` (✓) - Parse JSON với query support
|
| 43 |
+
- `yaml_parse` (✓) - Parse YAML
|
| 44 |
+
- `csv_parse` (✓) - Parse CSV
|
| 45 |
+
|
| 46 |
+
### SYSTEM
|
| 47 |
+
- `datetime` (✓) - DateTime operations + timezone
|
| 48 |
+
|
| 49 |
+
### NETWORK
|
| 50 |
+
- `dns_lookup` (✓) - DNS lookup (A, AAAA, MX, NS, CNAME, TXT)
|
| 51 |
+
- `ping` (✓) - Ping host
|
| 52 |
+
|
| 53 |
+
### CRYPTO
|
| 54 |
+
- `hash` (✓) - Compute hash (md5, sha1, sha256, sha512, blake2)
|
| 55 |
+
- `encrypt` (⚡) - AES-256-GCM encrypt/decrypt
|
| 56 |
+
|
| 57 |
+
### FILE (Archive)
|
| 58 |
+
- `archive` (⚠) - ZIP/TAR create/extract/list
|
| 59 |
+
|
| 60 |
+
## Usage
|
| 61 |
+
|
| 62 |
+
```python
|
| 63 |
+
from nexus.tools import get_global_registry, ToolContext
|
| 64 |
+
|
| 65 |
+
registry = get_global_registry()
|
| 66 |
+
|
| 67 |
+
# List all tools
|
| 68 |
+
print(registry.list_tools())
|
| 69 |
+
|
| 70 |
+
# Execute tool
|
| 71 |
+
from nexus.tools.base import ToolContext
|
| 72 |
+
ctx = ToolContext(working_dir="/tmp")
|
| 73 |
+
result = registry.execute("file_read", {"path": "/etc/hostname"}, ctx)
|
| 74 |
+
print(result.output)
|
| 75 |
+
|
| 76 |
+
# Check safety
|
| 77 |
+
tool = registry.get("file_delete")
|
| 78 |
+
print(f"Safety: {tool.safety.value}")
|
| 79 |
+
```
|
| 80 |
+
|
| 81 |
+
## Audit Log
|
| 82 |
+
|
| 83 |
+
All tool calls are logged to `./logs/tool_audit.jsonl`:
|
| 84 |
+
|
| 85 |
+
```json
|
| 86 |
+
{
|
| 87 |
+
"timestamp": 1234567890.123,
|
| 88 |
+
"tool": "file_write",
|
| 89 |
+
"safety": "moderate",
|
| 90 |
+
"args": {"path": "/tmp/test.txt", "content": "hello"},
|
| 91 |
+
"working_dir": ".",
|
| 92 |
+
"user_id": null,
|
| 93 |
+
"success": true,
|
| 94 |
+
"return_code": 0,
|
| 95 |
+
"duration": 0.001
|
| 96 |
+
}
|
| 97 |
+
```
|
| 98 |
+
|
| 99 |
+
## Safety Features
|
| 100 |
+
|
| 101 |
+
1. **Confirmation required** for DANGEROUS and DESTRUCTIVE tools
|
| 102 |
+
2. **Dry-run mode** to preview actions without executing
|
| 103 |
+
3. **Pre-hooks** for rate limiting, auth checks
|
| 104 |
+
4. **Post-hooks** for metrics, notifications
|
| 105 |
+
5. **Audit log** for compliance
|
| 106 |
+
6. **Blocked commands** for known dangerous patterns
|
| 107 |
+
|
| 108 |
+
## Custom Tools
|
| 109 |
+
|
| 110 |
+
```python
|
| 111 |
+
from nexus.tools.base import Tool, ToolResult, ToolContext, ToolCategory, ToolSafety
|
| 112 |
+
|
| 113 |
+
class MyTool(Tool):
|
| 114 |
+
category = ToolCategory.FILE
|
| 115 |
+
safety = ToolSafety.SAFE
|
| 116 |
+
|
| 117 |
+
@property
|
| 118 |
+
def name(self) -> str:
|
| 119 |
+
return "my_tool"
|
| 120 |
+
|
| 121 |
+
@property
|
| 122 |
+
def description(self) -> str:
|
| 123 |
+
return "My custom tool"
|
| 124 |
+
|
| 125 |
+
@property
|
| 126 |
+
def parameters(self) -> dict:
|
| 127 |
+
return {
|
| 128 |
+
"type": "object",
|
| 129 |
+
"properties": {"input": {"type": "string"}},
|
| 130 |
+
"required": ["input"],
|
| 131 |
+
}
|
| 132 |
+
|
| 133 |
+
def execute(self, args, context):
|
| 134 |
+
return ToolResult(
|
| 135 |
+
success=True,
|
| 136 |
+
output=f"Processed: {args['input']}",
|
| 137 |
+
)
|
| 138 |
+
|
| 139 |
+
# Register
|
| 140 |
+
from nexus.tools import get_global_registry
|
| 141 |
+
get_global_registry().register(MyTool())
|
| 142 |
+
```
|
| 143 |
+
|
| 144 |
+
## Tool Calling via Natural Language
|
| 145 |
+
|
| 146 |
+
Agent có thể detect tool calls từ natural language:
|
| 147 |
+
|
| 148 |
+
- "read file /etc/hostname" → `file_read`
|
| 149 |
+
- "run ls -la" → `shell_exec`
|
| 150 |
+
- "search for TODO in src/" → `regex_search`
|
| 151 |
+
- "fetch https://example.com" → `web_fetch`
|
| 152 |
+
- "git status" → `git_ops`
|
| 153 |
+
|
| 154 |
+
Hoặc JSON format:
|
| 155 |
+
```json
|
| 156 |
+
{"tool": "file_read", "args": {"path": "/etc/hostname"}}
|
| 157 |
+
```
|
| 158 |
+
|
| 159 |
+
Or @-mention:
|
| 160 |
+
```
|
| 161 |
+
@file_read path=/etc/hostname
|
| 162 |
+
```
|
docs/TRAINING.md
ADDED
|
@@ -0,0 +1,77 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Huấn luyện Nexus Coder / Training Nexus Coder
|
| 2 |
+
|
| 3 |
+
## Tổng quan / Overview
|
| 4 |
+
|
| 5 |
+
Nexus Coder v0.1 có thể được huấn luyện với script `scripts/train.py`. Training data được "hardcoded" với thông tin tác giả.
|
| 6 |
+
|
| 7 |
+
## Training Data
|
| 8 |
+
|
| 9 |
+
Dữ liệu huấn luyện nằm trong `nexus/training/dataset.py` và chứa:
|
| 10 |
+
- Q&A về tác giả (Hieu Louis)
|
| 11 |
+
- Sample code snippets
|
| 12 |
+
- Small talk examples
|
| 13 |
+
- Cả tiếng Việt và tiếng Anh
|
| 14 |
+
|
| 15 |
+
Để thêm dữ liệu, chỉnh sửa `AUTHOR_TRAINING_DATA` trong file đó.
|
| 16 |
+
|
| 17 |
+
## Cấu hình / Configuration
|
| 18 |
+
|
| 19 |
+
### Tiny config (CPU)
|
| 20 |
+
```bash
|
| 21 |
+
python scripts/train.py --steps 100 --batch_size 2 --max_length 64
|
| 22 |
+
```
|
| 23 |
+
|
| 24 |
+
### Full 10B config (cần GPU)
|
| 25 |
+
```bash
|
| 26 |
+
python scripts/train.py --full --steps 5000 --batch_size 4
|
| 27 |
+
```
|
| 28 |
+
|
| 29 |
+
## Yêu cầu hệ thống / System Requirements
|
| 30 |
+
|
| 31 |
+
### Tiny config
|
| 32 |
+
- CPU: bất kỳ
|
| 33 |
+
- RAM: 2GB+
|
| 34 |
+
- Disk: 100MB
|
| 35 |
+
|
| 36 |
+
### Full 10B config
|
| 37 |
+
- GPU: cần nhiều GPU (VD: 4x A100 80GB)
|
| 38 |
+
- RAM: 64GB+
|
| 39 |
+
- Disk: 50GB+ cho checkpoints
|
| 40 |
+
- Training time: nhiều ngày/tuần
|
| 41 |
+
|
| 42 |
+
## Hyperparameters mặc định
|
| 43 |
+
|
| 44 |
+
| Tham số | Giá trị |
|
| 45 |
+
|---------|---------|
|
| 46 |
+
| Learning rate | 5e-4 |
|
| 47 |
+
| Weight decay | 0.01 |
|
| 48 |
+
| Warmup steps | 100 |
|
| 49 |
+
| Max steps | 5000 |
|
| 50 |
+
| Batch size | 4 |
|
| 51 |
+
| Gradient accumulation | 4 |
|
| 52 |
+
| Save steps | 500 |
|
| 53 |
+
| Max grad norm | 1.0 |
|
| 54 |
+
| LR schedule | Cosine |
|
| 55 |
+
| Optimizer | AdamW (β1=0.9, β2=0.95) |
|
| 56 |
+
|
| 57 |
+
## Outputs
|
| 58 |
+
|
| 59 |
+
Training sẽ tạo:
|
| 60 |
+
- `checkpoints/nexus_coder-step-{N}.pt` - checkpoint
|
| 61 |
+
- `checkpoints/nexus_coder-final.pt` - final checkpoint
|
| 62 |
+
- `checkpoints/tokenizer.json` - trained tokenizer
|
| 63 |
+
- `checkpoints/training_log.json` - training log
|
| 64 |
+
|
| 65 |
+
## Tiếp tục từ checkpoint
|
| 66 |
+
|
| 67 |
+
```bash
|
| 68 |
+
# Đang cập nhật trong v0.2
|
| 69 |
+
```
|
| 70 |
+
|
| 71 |
+
## Lưu ý / Notes
|
| 72 |
+
|
| 73 |
+
⚠️ **v0.1 chỉ là foundation**:
|
| 74 |
+
- Tiny config chỉ dùng để verify code chạy được
|
| 75 |
+
- Full 10B cần GPU nhiều VRAM và nhiều thời gian
|
| 76 |
+
- Model chưa được pre-trained trên corpus lớn
|
| 77 |
+
- Để model trả lời thực sự, cần train thêm nhiều dữ liệu
|
nexus/__init__.py
ADDED
|
@@ -0,0 +1,81 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
Nexus Coder - Super CyberGym AI
|
| 3 |
+
================================
|
| 4 |
+
v0.4.0 - CyberForge edition
|
| 5 |
+
|
| 6 |
+
Model AI được tạo bởi Hieu Louis (2026)
|
| 7 |
+
|
| 8 |
+
Tổng tham số: 423 tỷ (423B)
|
| 9 |
+
Tham số kích hoạt: 39 tỷ (39B active)
|
| 10 |
+
Cửa sổ ngữ cảnh: 3,000,000 tokens (3M)
|
| 11 |
+
|
| 12 |
+
Kiến trúc: CyberForge MoE Transformer
|
| 13 |
+
- GQA + RoPE (YaRN-scaled) + RMSNorm + SwiGLU + FlashAttention-2
|
| 14 |
+
- Sliding Window Attention + QK-norm + KV cache quantization
|
| 15 |
+
- MLP-parallel + Gradient checkpointing
|
| 16 |
+
- CyberGym training: Mutation Pressure + Code Genome + Expert Speciation + CEP
|
| 17 |
+
|
| 18 |
+
Skills: 60+ · Tools: 80+ · Data sources: 8+ · Code corpus: 3000+ repos
|
| 19 |
+
|
| 20 |
+
Tác giả: Hieu Louis
|
| 21 |
+
GitHub: mhieuhonda
|
| 22 |
+
Năm: 2026
|
| 23 |
+
"""
|
| 24 |
+
|
| 25 |
+
__version__ = "0.4.0"
|
| 26 |
+
__author__ = "Hieu Louis"
|
| 27 |
+
__github__ = "mhieuhonda"
|
| 28 |
+
__year__ = "2026"
|
| 29 |
+
__license__ = "NexusCoder Attribution License v1.0"
|
| 30 |
+
|
| 31 |
+
# Thông tin tác giả được "huấn luyện cứng" vào model
|
| 32 |
+
AUTHOR_INFO = {
|
| 33 |
+
"name": "Hieu Louis",
|
| 34 |
+
"github": "mhieuhonda",
|
| 35 |
+
"year": "2026",
|
| 36 |
+
"description": (
|
| 37 |
+
"Nexus Coder là dự án AI cá nhân do Hieu Louis tự xây dựng từ đầu "
|
| 38 |
+
"với kiến trúc CyberForge MoE tiên tiến, kết hợp CyberGym training."
|
| 39 |
+
),
|
| 40 |
+
"model_name": "Nexus Coder",
|
| 41 |
+
"agent_name": "Nexus",
|
| 42 |
+
"version": "0.4.0",
|
| 43 |
+
"architecture": (
|
| 44 |
+
"CyberForge MoE Transformer (GQA + RoPE/YaRN + RMSNorm + SwiGLU + "
|
| 45 |
+
"FlashAttention-2 + Sliding Window + QK-norm + KV-cache quant + "
|
| 46 |
+
"MLP-parallel + Gradient checkpointing)"
|
| 47 |
+
),
|
| 48 |
+
"total_params": "~423B (variants: 5M tiny → 423B)",
|
| 49 |
+
"active_params": "~39B (variants: 2M tiny → 39B)",
|
| 50 |
+
"context_window": "3,000,000 tokens (3M, via YaRN + CEP)",
|
| 51 |
+
"python_version": "3.12.13",
|
| 52 |
+
"skills_count": "60+",
|
| 53 |
+
"tools_count": "80+",
|
| 54 |
+
"data_sources": "8+ (GitHub 3000+ repos, HuggingFace, arXiv, Wikipedia, StackOverflow, The-Stack, StarCoder2-data, Python-Alpaca)",
|
| 55 |
+
"training_methodology": "CyberForge (Mutation Pressure Training + Code Genome Init + Expert Speciation + Context Expansion Protocol)",
|
| 56 |
+
"training_frameworks_referenced": "litgpt, LlamaFactory, axolotl, OpenHands, omp-gym",
|
| 57 |
+
}
|
| 58 |
+
|
| 59 |
+
|
| 60 |
+
# Lazy import để giảm startup time
|
| 61 |
+
def __getattr__(name: str):
|
| 62 |
+
if name == "NexusConfig":
|
| 63 |
+
from .config import NexusConfig
|
| 64 |
+
return NexusConfig
|
| 65 |
+
if name == "NEXUS_CODER_10B_CONFIG":
|
| 66 |
+
from .config import NEXUS_CODER_10B_CONFIG
|
| 67 |
+
return NEXUS_CODER_10B_CONFIG
|
| 68 |
+
if name == "NEXUS_CODER_423B_CONFIG":
|
| 69 |
+
from .config import NEXUS_CODER_423B_CONFIG
|
| 70 |
+
return NEXUS_CODER_423B_CONFIG
|
| 71 |
+
raise AttributeError(f"module 'nexus' has no attribute {name!r}")
|
| 72 |
+
|
| 73 |
+
|
| 74 |
+
__all__ = [
|
| 75 |
+
"AUTHOR_INFO",
|
| 76 |
+
"__version__",
|
| 77 |
+
"__author__",
|
| 78 |
+
"__github__",
|
| 79 |
+
"__year__",
|
| 80 |
+
"__license__",
|
| 81 |
+
]
|
nexus/agent/__init__.py
ADDED
|
@@ -0,0 +1,4 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Agent package."""
|
| 2 |
+
from .agent import NexusAgent
|
| 3 |
+
|
| 4 |
+
__all__ = ["NexusAgent"]
|
nexus/agent/agent.py
ADDED
|
@@ -0,0 +1,364 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
Nexus Agent v0.2 - AI Agent với Skills + Tools
|
| 3 |
+
===============================================
|
| 4 |
+
Major upgrade từ v0.1:
|
| 5 |
+
- Tích hợp SkillRegistry (15+ skills)
|
| 6 |
+
- Tích hợp ToolRegistry (15+ tools)
|
| 7 |
+
- Memory system
|
| 8 |
+
- Planner cho multi-step tasks
|
| 9 |
+
- Tool routing thông minh
|
| 10 |
+
- Safety guardrails
|
| 11 |
+
- Audit logging
|
| 12 |
+
"""
|
| 13 |
+
from __future__ import annotations
|
| 14 |
+
|
| 15 |
+
from typing import Optional, List, Dict, Any, Callable
|
| 16 |
+
import json
|
| 17 |
+
import os
|
| 18 |
+
from datetime import datetime
|
| 19 |
+
from pathlib import Path
|
| 20 |
+
|
| 21 |
+
from ..config import NexusConfig
|
| 22 |
+
from ..inference.generator import NexusGenerator, DEFAULT_SYSTEM_PROMPT
|
| 23 |
+
from ..tokenizer.tokenizer import NexusTokenizer
|
| 24 |
+
from ..model.nexus_coder import NexusCoderForCausalLM
|
| 25 |
+
from .. import AUTHOR_INFO
|
| 26 |
+
from ..skills import SkillRegistry, get_global_registry as get_skill_registry
|
| 27 |
+
from ..tools import ToolRegistry, ToolContext, get_global_registry as get_tool_registry
|
| 28 |
+
from ..safety import SafetyFilter, get_default_guardrails
|
| 29 |
+
from .memory import ConversationMemory
|
| 30 |
+
from .planner import TaskPlanner
|
| 31 |
+
from .router import ToolRouter
|
| 32 |
+
|
| 33 |
+
|
| 34 |
+
class NexusAgent:
|
| 35 |
+
"""AI Agent v0.2 - Wrapper cấp cao cho Nexus Coder.
|
| 36 |
+
|
| 37 |
+
Features:
|
| 38 |
+
- Skill-based routing (15+ skills)
|
| 39 |
+
- Tool use (15+ tools)
|
| 40 |
+
- Conversation memory
|
| 41 |
+
- Task planning
|
| 42 |
+
- Safety guardrails
|
| 43 |
+
- Audit logging
|
| 44 |
+
|
| 45 |
+
Usage:
|
| 46 |
+
agent = NexusAgent()
|
| 47 |
+
agent.chat() # Interactive
|
| 48 |
+
# or
|
| 49 |
+
response = agent.respond("Viết hàm fibonacci")
|
| 50 |
+
"""
|
| 51 |
+
|
| 52 |
+
def __init__(
|
| 53 |
+
self,
|
| 54 |
+
generator: Optional[NexusGenerator] = None,
|
| 55 |
+
config: Optional[NexusConfig] = None,
|
| 56 |
+
name: str = "Nexus",
|
| 57 |
+
personality: str = "humorous",
|
| 58 |
+
language: str = "bilingual",
|
| 59 |
+
enable_logging: bool = True,
|
| 60 |
+
log_dir: str = "./logs",
|
| 61 |
+
enable_skills: bool = True,
|
| 62 |
+
enable_tools: bool = True,
|
| 63 |
+
enable_memory: bool = True,
|
| 64 |
+
enable_planner: bool = True,
|
| 65 |
+
enable_safety: bool = True,
|
| 66 |
+
working_dir: str = ".",
|
| 67 |
+
):
|
| 68 |
+
self.config = config or NexusConfig()
|
| 69 |
+
self.name = name
|
| 70 |
+
self.personality = personality
|
| 71 |
+
self.language = language
|
| 72 |
+
self.author_info = AUTHOR_INFO
|
| 73 |
+
self.working_dir = working_dir
|
| 74 |
+
|
| 75 |
+
# Generator (model + tokenizer)
|
| 76 |
+
if generator is None:
|
| 77 |
+
self.generator = NexusGenerator(
|
| 78 |
+
model=NexusCoderForCausalLM(self.config),
|
| 79 |
+
tokenizer=NexusTokenizer(),
|
| 80 |
+
config=self.config,
|
| 81 |
+
)
|
| 82 |
+
else:
|
| 83 |
+
self.generator = generator
|
| 84 |
+
|
| 85 |
+
# Logging
|
| 86 |
+
self.enable_logging = enable_logging
|
| 87 |
+
self.log_dir = log_dir
|
| 88 |
+
if enable_logging:
|
| 89 |
+
os.makedirs(log_dir, exist_ok=True)
|
| 90 |
+
|
| 91 |
+
# Skills (v0.2 NEW)
|
| 92 |
+
self.enable_skills = enable_skills and self.config.enable_skills
|
| 93 |
+
self.skill_registry: Optional[SkillRegistry] = (
|
| 94 |
+
get_skill_registry() if self.enable_skills else None
|
| 95 |
+
)
|
| 96 |
+
|
| 97 |
+
# Tools (v0.2 NEW)
|
| 98 |
+
self.enable_tools = enable_tools and self.config.enable_tools
|
| 99 |
+
self.tool_registry: Optional[ToolRegistry] = (
|
| 100 |
+
get_tool_registry() if self.enable_tools else None
|
| 101 |
+
)
|
| 102 |
+
self.tool_router = ToolRouter(self.tool_registry) if self.tool_registry else None
|
| 103 |
+
|
| 104 |
+
# Memory (v0.2 NEW)
|
| 105 |
+
self.enable_memory = enable_memory and self.config.enable_memory
|
| 106 |
+
self.memory = ConversationMemory() if self.enable_memory else None
|
| 107 |
+
|
| 108 |
+
# Planner (v0.2 NEW)
|
| 109 |
+
self.enable_planner = enable_planner and self.config.enable_planner
|
| 110 |
+
self.planner = TaskPlanner() if self.enable_planner else None
|
| 111 |
+
|
| 112 |
+
# Safety (v0.2 NEW)
|
| 113 |
+
self.enable_safety = enable_safety and self.config.enable_safety_filter
|
| 114 |
+
self.safety_filter = SafetyFilter() if self.enable_safety else None
|
| 115 |
+
self.guardrails = get_default_guardrails() if self.enable_safety else None
|
| 116 |
+
|
| 117 |
+
# Stats
|
| 118 |
+
self._stats = {
|
| 119 |
+
"total_messages": 0,
|
| 120 |
+
"skills_used": 0,
|
| 121 |
+
"tools_called": 0,
|
| 122 |
+
"safety_blocks": 0,
|
| 123 |
+
"session_start": datetime.now().isoformat(),
|
| 124 |
+
}
|
| 125 |
+
|
| 126 |
+
print(f"✓ Nexus Agent v0.2 initialized")
|
| 127 |
+
print(f" Tên: {self.name}")
|
| 128 |
+
print(f" Tác giả: {self.author_info['name']}")
|
| 129 |
+
print(f" Phiên bản: {self.author_info['version']}")
|
| 130 |
+
print(f" Skills: {len(self.skill_registry) if self.skill_registry else 0}")
|
| 131 |
+
print(f" Tools: {len(self.tool_registry) if self.tool_registry else 0}")
|
| 132 |
+
print(f" Memory: {'✓' if self.memory else '✗'}")
|
| 133 |
+
print(f" Planner: {'✓' if self.planner else '✗'}")
|
| 134 |
+
print(f" Safety: {'✓' if self.safety_filter else '✗'}")
|
| 135 |
+
|
| 136 |
+
def respond(self, user_input: str, **kwargs) -> str:
|
| 137 |
+
"""Phản hồi tin nhắn từ người dùng."""
|
| 138 |
+
start_time = datetime.now()
|
| 139 |
+
self._stats["total_messages"] += 1
|
| 140 |
+
|
| 141 |
+
# Safety check (input)
|
| 142 |
+
if self.guardrails:
|
| 143 |
+
guard_result = self.guardrails.check(user_input)
|
| 144 |
+
if not guard_result["allowed"]:
|
| 145 |
+
self._stats["safety_blocks"] += 1
|
| 146 |
+
return f"⚠️ {guard_result['message']}"
|
| 147 |
+
|
| 148 |
+
# Add to memory
|
| 149 |
+
if self.memory:
|
| 150 |
+
self.memory.add(role="user", content=user_input)
|
| 151 |
+
|
| 152 |
+
# Try skill routing
|
| 153 |
+
skill_used = None
|
| 154 |
+
skill_result = None
|
| 155 |
+
if self.skill_registry:
|
| 156 |
+
from ..skills.base import SkillContext
|
| 157 |
+
ctx = SkillContext(
|
| 158 |
+
prompt=user_input,
|
| 159 |
+
history=self.memory.get_history() if self.memory else [],
|
| 160 |
+
**kwargs,
|
| 161 |
+
)
|
| 162 |
+
skill = self.skill_registry.route(user_input, ctx)
|
| 163 |
+
if skill:
|
| 164 |
+
skill_used = skill.name
|
| 165 |
+
skill_result = skill.execute(ctx)
|
| 166 |
+
self._stats["skills_used"] += 1
|
| 167 |
+
|
| 168 |
+
# Check for tool calls in user input
|
| 169 |
+
tool_calls_made = []
|
| 170 |
+
if self.tool_router:
|
| 171 |
+
tool_calls = self.tool_router.detect_tool_calls(user_input)
|
| 172 |
+
for tc in tool_calls[:self.config.max_tool_calls]:
|
| 173 |
+
result = self.tool_registry.execute(
|
| 174 |
+
tc["name"],
|
| 175 |
+
tc.get("args", {}),
|
| 176 |
+
ToolContext(working_dir=self.working_dir),
|
| 177 |
+
)
|
| 178 |
+
tool_calls_made.append({
|
| 179 |
+
"tool": tc["name"],
|
| 180 |
+
"success": result.success,
|
| 181 |
+
"output": result.output[:500] if result.output else "",
|
| 182 |
+
})
|
| 183 |
+
self._stats["tools_called"] += 1
|
| 184 |
+
|
| 185 |
+
# Generate response
|
| 186 |
+
try:
|
| 187 |
+
# Build enhanced prompt with skill/tool context
|
| 188 |
+
enhanced_input = user_input
|
| 189 |
+
if skill_result:
|
| 190 |
+
enhanced_input += f"\n\n[Skill: {skill_used}] {skill_result.output}"
|
| 191 |
+
if tool_calls_made:
|
| 192 |
+
enhanced_input += "\n\n[Tool results:]"
|
| 193 |
+
for tc in tool_calls_made:
|
| 194 |
+
enhanced_input += f"\n- {tc['tool']}: {tc['output'][:200]}"
|
| 195 |
+
|
| 196 |
+
response = self.generator.chat(enhanced_input, **kwargs)
|
| 197 |
+
except Exception as e:
|
| 198 |
+
response = f"⚠️ Xin lỗi, có lỗi xảy ra: {e}"
|
| 199 |
+
|
| 200 |
+
# Add to memory
|
| 201 |
+
if self.memory:
|
| 202 |
+
self.memory.add(role="assistant", content=response)
|
| 203 |
+
|
| 204 |
+
elapsed = (datetime.now() - start_time).total_seconds()
|
| 205 |
+
|
| 206 |
+
# Logging
|
| 207 |
+
if self.enable_logging:
|
| 208 |
+
self._log_interaction(
|
| 209 |
+
user_input=user_input,
|
| 210 |
+
response=response,
|
| 211 |
+
elapsed=elapsed,
|
| 212 |
+
skill_used=skill_used,
|
| 213 |
+
tools_used=[t["tool"] for t in tool_calls_made],
|
| 214 |
+
)
|
| 215 |
+
|
| 216 |
+
return response
|
| 217 |
+
|
| 218 |
+
def chat(self) -> None:
|
| 219 |
+
"""Bắt đầu chế độ chat tương tác."""
|
| 220 |
+
print("\n" + "=" * 70)
|
| 221 |
+
print(f" 🤖 {self.name} Agent v0.2.0")
|
| 222 |
+
print(f" Tác giả: {self.author_info['name']}")
|
| 223 |
+
print(f" Phiên bản: {self.author_info['version']}")
|
| 224 |
+
print(f" Ngôn ngữ: {'Song ngữ' if self.language == 'bilingual' else self.language}")
|
| 225 |
+
print(f" Skills: {len(self.skill_registry) if self.skill_registry else 0}")
|
| 226 |
+
print(f" Tools: {len(self.tool_registry) if self.tool_registry else 0}")
|
| 227 |
+
print("=" * 70)
|
| 228 |
+
print("Commands:")
|
| 229 |
+
print(" exit/quit - Thoát")
|
| 230 |
+
print(" reset - Xóa lịch sử")
|
| 231 |
+
print(" info - Thông tin model")
|
| 232 |
+
print(" skills - Liệt kê skills")
|
| 233 |
+
print(" tools - Liệt kê tools")
|
| 234 |
+
print(" stats - Thống kê session")
|
| 235 |
+
print("-" * 70 + "\n")
|
| 236 |
+
|
| 237 |
+
while True:
|
| 238 |
+
try:
|
| 239 |
+
user_input = input("\n🧑 Bạn: ").strip()
|
| 240 |
+
except (EOFError, KeyboardInterrupt):
|
| 241 |
+
print("\n\n👋 Tạm biệt!")
|
| 242 |
+
break
|
| 243 |
+
|
| 244 |
+
if not user_input:
|
| 245 |
+
continue
|
| 246 |
+
|
| 247 |
+
cmd = user_input.lower()
|
| 248 |
+
if cmd in ["exit", "quit"]:
|
| 249 |
+
print(f"\n👋 Tạm biệt! Hẹn gặp lại bạn. - {self.name}")
|
| 250 |
+
break
|
| 251 |
+
elif cmd == "reset":
|
| 252 |
+
if self.memory:
|
| 253 |
+
self.memory.clear()
|
| 254 |
+
self.generator.reset_conversation()
|
| 255 |
+
print("\n🔄 Đã xóa lịch sử trò chuyện.")
|
| 256 |
+
continue
|
| 257 |
+
elif cmd == "info":
|
| 258 |
+
self._print_info()
|
| 259 |
+
continue
|
| 260 |
+
elif cmd == "skills":
|
| 261 |
+
self._print_skills()
|
| 262 |
+
continue
|
| 263 |
+
elif cmd == "tools":
|
| 264 |
+
self._print_tools()
|
| 265 |
+
continue
|
| 266 |
+
elif cmd == "stats":
|
| 267 |
+
self._print_stats()
|
| 268 |
+
continue
|
| 269 |
+
|
| 270 |
+
response = self.respond(user_input)
|
| 271 |
+
print(f"\n🤖 {self.name}: {response}")
|
| 272 |
+
|
| 273 |
+
def _print_info(self) -> None:
|
| 274 |
+
"""In thông tin về model."""
|
| 275 |
+
stats = self.config.estimated_total_params()
|
| 276 |
+
print("\n" + "=" * 60)
|
| 277 |
+
print(f" Model: {self.author_info['model_name']}")
|
| 278 |
+
print(f" Agent: {self.author_info['agent_name']}")
|
| 279 |
+
print(f" Version: {self.author_info['version']}")
|
| 280 |
+
print(f" Tác giả: {self.author_info['name']}")
|
| 281 |
+
print(f" GitHub: {self.author_info['github']}")
|
| 282 |
+
print("-" * 60)
|
| 283 |
+
print(f" Tổng tham số: {stats['total_params_billion']:.2f}B")
|
| 284 |
+
print(f" Tham số active: {stats['active_params_billion']:.2f}B")
|
| 285 |
+
print(f" Context window: {self.config.max_position_embeddings:,} tokens")
|
| 286 |
+
print(f" Experts: {self.config.num_experts} (active: {self.config.num_active_experts})")
|
| 287 |
+
print(f" Python: 3.12.13")
|
| 288 |
+
print("=" * 60)
|
| 289 |
+
|
| 290 |
+
def _print_skills(self) -> None:
|
| 291 |
+
"""Liệt kê skills."""
|
| 292 |
+
if not self.skill_registry:
|
| 293 |
+
print("\n❌ Skills chưa được enable")
|
| 294 |
+
return
|
| 295 |
+
print("\n" + "=" * 60)
|
| 296 |
+
print(" Available Skills")
|
| 297 |
+
print("=" * 60)
|
| 298 |
+
by_cat = self.skill_registry.list_by_category()
|
| 299 |
+
for cat, skills in sorted(by_cat.items()):
|
| 300 |
+
print(f"\n [{cat.upper()}]")
|
| 301 |
+
for s in skills:
|
| 302 |
+
skill = self.skill_registry.get(s)
|
| 303 |
+
print(f" • {s}: {skill.description}")
|
| 304 |
+
print("\n" + "=" * 60)
|
| 305 |
+
|
| 306 |
+
def _print_tools(self) -> None:
|
| 307 |
+
"""Liệt kê tools."""
|
| 308 |
+
if not self.tool_registry:
|
| 309 |
+
print("\n❌ Tools chưa được enable")
|
| 310 |
+
return
|
| 311 |
+
print("\n" + "=" * 60)
|
| 312 |
+
print(" Available Tools")
|
| 313 |
+
print("=" * 60)
|
| 314 |
+
by_cat = self.tool_registry.list_by_category()
|
| 315 |
+
for cat, tools in sorted(by_cat.items()):
|
| 316 |
+
print(f"\n [{cat.upper()}]")
|
| 317 |
+
for t in tools:
|
| 318 |
+
tool = self.tool_registry.get(t)
|
| 319 |
+
safety_icon = {
|
| 320 |
+
"safe": "✓", "moderate": "⚠", "dangerous": "⚡", "destructive": "💀"
|
| 321 |
+
}.get(tool.safety.value, "?")
|
| 322 |
+
print(f" {safety_icon} {t}: {tool.description}")
|
| 323 |
+
print("\n" + "=" * 60)
|
| 324 |
+
|
| 325 |
+
def _print_stats(self) -> None:
|
| 326 |
+
"""In thống kê session."""
|
| 327 |
+
print("\n" + "=" * 60)
|
| 328 |
+
print(" Session Stats")
|
| 329 |
+
print("=" * 60)
|
| 330 |
+
for k, v in self._stats.items():
|
| 331 |
+
print(f" {k}: {v}")
|
| 332 |
+
print("=" * 60)
|
| 333 |
+
|
| 334 |
+
def _log_interaction(
|
| 335 |
+
self,
|
| 336 |
+
user_input: str,
|
| 337 |
+
response: str,
|
| 338 |
+
elapsed: float,
|
| 339 |
+
skill_used: Optional[str] = None,
|
| 340 |
+
tools_used: Optional[List[str]] = None,
|
| 341 |
+
) -> None:
|
| 342 |
+
"""Log tương tác vào file."""
|
| 343 |
+
log_file = os.path.join(self.log_dir, f"chat_{datetime.now().strftime('%Y%m%d')}.jsonl")
|
| 344 |
+
entry = {
|
| 345 |
+
"timestamp": datetime.now().isoformat(),
|
| 346 |
+
"user": user_input,
|
| 347 |
+
"assistant": response,
|
| 348 |
+
"elapsed_seconds": elapsed,
|
| 349 |
+
"skill_used": skill_used,
|
| 350 |
+
"tools_used": tools_used or [],
|
| 351 |
+
}
|
| 352 |
+
try:
|
| 353 |
+
with open(log_file, "a", encoding="utf-8") as f:
|
| 354 |
+
f.write(json.dumps(entry, ensure_ascii=False) + "\n")
|
| 355 |
+
except Exception:
|
| 356 |
+
pass
|
| 357 |
+
|
| 358 |
+
def get_author_info(self) -> Dict[str, str]:
|
| 359 |
+
"""Trả về thông tin tác giả."""
|
| 360 |
+
return self.author_info
|
| 361 |
+
|
| 362 |
+
def get_stats(self) -> Dict[str, Any]:
|
| 363 |
+
"""Trả về stats."""
|
| 364 |
+
return dict(self._stats)
|
nexus/agent/memory.py
ADDED
|
@@ -0,0 +1,191 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Memory System - Quản lý lịch sử hội thoại."""
|
| 2 |
+
from __future__ import annotations
|
| 3 |
+
|
| 4 |
+
from typing import List, Dict, Any, Optional
|
| 5 |
+
from dataclasses import dataclass, field
|
| 6 |
+
from datetime import datetime
|
| 7 |
+
import json
|
| 8 |
+
|
| 9 |
+
|
| 10 |
+
@dataclass
|
| 11 |
+
class Message:
|
| 12 |
+
"""Một message trong hội thoại."""
|
| 13 |
+
role: str # "system", "user", "assistant", "tool"
|
| 14 |
+
content: str
|
| 15 |
+
timestamp: str = field(default_factory=lambda: datetime.now().isoformat())
|
| 16 |
+
metadata: Dict[str, Any] = field(default_factory=dict)
|
| 17 |
+
|
| 18 |
+
|
| 19 |
+
class ConversationMemory:
|
| 20 |
+
"""Quản lý lịch sử hội thoại với sliding window.
|
| 21 |
+
|
| 22 |
+
Features:
|
| 23 |
+
- Lưu trữ messages
|
| 24 |
+
- Sliding window (giữ N messages gần nhất)
|
| 25 |
+
- Summarization (khi đầy, summarize cũ)
|
| 26 |
+
- Importance scoring
|
| 27 |
+
- Search trong history
|
| 28 |
+
|
| 29 |
+
Usage:
|
| 30 |
+
memory = ConversationMemory(max_messages=50)
|
| 31 |
+
memory.add(role="user", content="Hello")
|
| 32 |
+
memory.add(role="assistant", content="Hi there!")
|
| 33 |
+
history = memory.get_history()
|
| 34 |
+
"""
|
| 35 |
+
|
| 36 |
+
def __init__(
|
| 37 |
+
self,
|
| 38 |
+
max_messages: int = 50,
|
| 39 |
+
max_tokens: int = 4000,
|
| 40 |
+
summarize_threshold: float = 0.8,
|
| 41 |
+
):
|
| 42 |
+
self.max_messages = max_messages
|
| 43 |
+
self.max_tokens = max_tokens
|
| 44 |
+
self.summarize_threshold = summarize_threshold
|
| 45 |
+
self._messages: List[Message] = []
|
| 46 |
+
self._summary: Optional[str] = None
|
| 47 |
+
self._importance_scores: List[float] = []
|
| 48 |
+
|
| 49 |
+
def add(
|
| 50 |
+
self,
|
| 51 |
+
role: str,
|
| 52 |
+
content: str,
|
| 53 |
+
metadata: Optional[Dict[str, Any]] = None,
|
| 54 |
+
importance: float = 0.5,
|
| 55 |
+
) -> None:
|
| 56 |
+
"""Add a message to memory."""
|
| 57 |
+
msg = Message(
|
| 58 |
+
role=role,
|
| 59 |
+
content=content,
|
| 60 |
+
metadata=metadata or {},
|
| 61 |
+
)
|
| 62 |
+
self._messages.append(msg)
|
| 63 |
+
self._importance_scores.append(importance)
|
| 64 |
+
|
| 65 |
+
# Trigger summarization if threshold reached
|
| 66 |
+
if len(self._messages) >= self.max_messages * self.summarize_threshold:
|
| 67 |
+
self._compress()
|
| 68 |
+
|
| 69 |
+
def get_history(
|
| 70 |
+
self,
|
| 71 |
+
last_n: Optional[int] = None,
|
| 72 |
+
include_summary: bool = True,
|
| 73 |
+
) -> List[Dict[str, str]]:
|
| 74 |
+
"""Get conversation history.
|
| 75 |
+
|
| 76 |
+
Args:
|
| 77 |
+
last_n: Only return last N messages (None = all)
|
| 78 |
+
include_summary: Include previous summary if available
|
| 79 |
+
|
| 80 |
+
Returns:
|
| 81 |
+
List of {"role": ..., "content": ...}
|
| 82 |
+
"""
|
| 83 |
+
history = []
|
| 84 |
+
if include_summary and self._summary:
|
| 85 |
+
history.append({
|
| 86 |
+
"role": "system",
|
| 87 |
+
"content": f"[Previous conversation summary]: {self._summary}",
|
| 88 |
+
})
|
| 89 |
+
|
| 90 |
+
messages = self._messages[-last_n:] if last_n else self._messages
|
| 91 |
+
for msg in messages:
|
| 92 |
+
history.append({
|
| 93 |
+
"role": msg.role,
|
| 94 |
+
"content": msg.content,
|
| 95 |
+
})
|
| 96 |
+
|
| 97 |
+
return history
|
| 98 |
+
|
| 99 |
+
def search(self, query: str, limit: int = 5) -> List[Dict[str, str]]:
|
| 100 |
+
"""Search in memory for relevant messages."""
|
| 101 |
+
query_lower = query.lower()
|
| 102 |
+
scored = []
|
| 103 |
+
for msg, score in zip(self._messages, self._importance_scores):
|
| 104 |
+
content_lower = msg.content.lower()
|
| 105 |
+
# Simple keyword matching
|
| 106 |
+
matches = sum(1 for word in query_lower.split() if word in content_lower)
|
| 107 |
+
if matches > 0:
|
| 108 |
+
relevance = matches / max(len(query_lower.split()), 1)
|
| 109 |
+
scored.append((relevance * score, msg))
|
| 110 |
+
|
| 111 |
+
scored.sort(key=lambda x: -x[0])
|
| 112 |
+
return [
|
| 113 |
+
{"role": m.role, "content": m.content}
|
| 114 |
+
for _, m in scored[:limit]
|
| 115 |
+
]
|
| 116 |
+
|
| 117 |
+
def clear(self) -> None:
|
| 118 |
+
"""Clear all memory."""
|
| 119 |
+
self._messages.clear()
|
| 120 |
+
self._importance_scores.clear()
|
| 121 |
+
self._summary = None
|
| 122 |
+
|
| 123 |
+
def _compress(self) -> None:
|
| 124 |
+
"""Compress old messages into summary."""
|
| 125 |
+
# Keep recent messages, summarize older ones
|
| 126 |
+
keep_count = self.max_messages // 2
|
| 127 |
+
old_messages = self._messages[:-keep_count]
|
| 128 |
+
old_scores = self._importance_scores[:-keep_count]
|
| 129 |
+
|
| 130 |
+
# Build summary (simple: concatenate key points)
|
| 131 |
+
summary_parts = []
|
| 132 |
+
for msg in old_messages:
|
| 133 |
+
if msg.role == "user":
|
| 134 |
+
summary_parts.append(f"User asked: {msg.content[:100]}")
|
| 135 |
+
elif msg.role == "assistant":
|
| 136 |
+
summary_parts.append(f"Assistant replied: {msg.content[:100]}")
|
| 137 |
+
|
| 138 |
+
new_summary = " | ".join(summary_parts[-10:]) # Last 10 interactions
|
| 139 |
+
|
| 140 |
+
if self._summary:
|
| 141 |
+
self._summary = f"{self._summary} | {new_summary}"
|
| 142 |
+
else:
|
| 143 |
+
self._summary = new_summary
|
| 144 |
+
|
| 145 |
+
# Truncate summary if too long
|
| 146 |
+
if len(self._summary) > 2000:
|
| 147 |
+
self._summary = self._summary[-2000:]
|
| 148 |
+
|
| 149 |
+
# Keep only recent messages
|
| 150 |
+
self._messages = self._messages[-keep_count:]
|
| 151 |
+
self._importance_scores = self._importance_scores[-keep_count:]
|
| 152 |
+
|
| 153 |
+
def stats(self) -> Dict[str, Any]:
|
| 154 |
+
"""Get memory stats."""
|
| 155 |
+
total_chars = sum(len(m.content) for m in self._messages)
|
| 156 |
+
return {
|
| 157 |
+
"message_count": len(self._messages),
|
| 158 |
+
"max_messages": self.max_messages,
|
| 159 |
+
"total_chars": total_chars,
|
| 160 |
+
"has_summary": self._summary is not None,
|
| 161 |
+
"summary_length": len(self._summary) if self._summary else 0,
|
| 162 |
+
}
|
| 163 |
+
|
| 164 |
+
def save(self, path: str) -> None:
|
| 165 |
+
"""Save memory to file."""
|
| 166 |
+
data = {
|
| 167 |
+
"messages": [
|
| 168 |
+
{"role": m.role, "content": m.content, "timestamp": m.timestamp, "metadata": m.metadata}
|
| 169 |
+
for m in self._messages
|
| 170 |
+
],
|
| 171 |
+
"summary": self._summary,
|
| 172 |
+
"max_messages": self.max_messages,
|
| 173 |
+
}
|
| 174 |
+
with open(path, "w", encoding="utf-8") as f:
|
| 175 |
+
json.dump(data, f, ensure_ascii=False, indent=2)
|
| 176 |
+
|
| 177 |
+
def load(self, path: str) -> None:
|
| 178 |
+
"""Load memory from file."""
|
| 179 |
+
with open(path, "r", encoding="utf-8") as f:
|
| 180 |
+
data = json.load(f)
|
| 181 |
+
self._messages = [
|
| 182 |
+
Message(
|
| 183 |
+
role=m["role"],
|
| 184 |
+
content=m["content"],
|
| 185 |
+
timestamp=m.get("timestamp", ""),
|
| 186 |
+
metadata=m.get("metadata", {}),
|
| 187 |
+
)
|
| 188 |
+
for m in data.get("messages", [])
|
| 189 |
+
]
|
| 190 |
+
self._summary = data.get("summary")
|
| 191 |
+
self.max_messages = data.get("max_messages", self.max_messages)
|
nexus/agent/planner.py
ADDED
|
@@ -0,0 +1,260 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Task Planner - Lập kế hoạch cho multi-step tasks."""
|
| 2 |
+
from __future__ import annotations
|
| 3 |
+
|
| 4 |
+
from typing import List, Dict, Any, Optional
|
| 5 |
+
from dataclasses import dataclass, field
|
| 6 |
+
from enum import Enum
|
| 7 |
+
|
| 8 |
+
|
| 9 |
+
class TaskStatus(str, Enum):
|
| 10 |
+
PENDING = "pending"
|
| 11 |
+
IN_PROGRESS = "in_progress"
|
| 12 |
+
COMPLETED = "completed"
|
| 13 |
+
FAILED = "failed"
|
| 14 |
+
SKIPPED = "skipped"
|
| 15 |
+
|
| 16 |
+
|
| 17 |
+
@dataclass
|
| 18 |
+
class Task:
|
| 19 |
+
"""Một task trong plan."""
|
| 20 |
+
id: int
|
| 21 |
+
description: str
|
| 22 |
+
skill: Optional[str] = None
|
| 23 |
+
tools: List[str] = field(default_factory=list)
|
| 24 |
+
depends_on: List[int] = field(default_factory=list)
|
| 25 |
+
status: TaskStatus = TaskStatus.PENDING
|
| 26 |
+
result: Optional[str] = None
|
| 27 |
+
metadata: Dict[str, Any] = field(default_factory=dict)
|
| 28 |
+
|
| 29 |
+
|
| 30 |
+
@dataclass
|
| 31 |
+
class Plan:
|
| 32 |
+
"""Một execution plan."""
|
| 33 |
+
id: str
|
| 34 |
+
goal: str
|
| 35 |
+
tasks: List[Task] = field(default_factory=list)
|
| 36 |
+
created_at: str = ""
|
| 37 |
+
status: TaskStatus = TaskStatus.PENDING
|
| 38 |
+
|
| 39 |
+
def add_task(self, task: Task) -> None:
|
| 40 |
+
self.tasks.append(task)
|
| 41 |
+
|
| 42 |
+
def get_next_task(self) -> Optional[Task]:
|
| 43 |
+
"""Get next pending task whose dependencies are met.
|
| 44 |
+
|
| 45 |
+
v0.4 fix: out-of-range dep IDs are treated as UNMET (not silently ignored).
|
| 46 |
+
"""
|
| 47 |
+
for task in self.tasks:
|
| 48 |
+
if task.status != TaskStatus.PENDING:
|
| 49 |
+
continue
|
| 50 |
+
# Check dependencies
|
| 51 |
+
deps_met = True
|
| 52 |
+
for dep_id in task.depends_on:
|
| 53 |
+
if dep_id < 0 or dep_id >= len(self.tasks):
|
| 54 |
+
# Invalid dep ID → mark unmet, do NOT silently pass
|
| 55 |
+
deps_met = False
|
| 56 |
+
break
|
| 57 |
+
if self.tasks[dep_id].status not in (TaskStatus.COMPLETED, TaskStatus.SKIPPED):
|
| 58 |
+
deps_met = False
|
| 59 |
+
break
|
| 60 |
+
if deps_met:
|
| 61 |
+
return task
|
| 62 |
+
return None
|
| 63 |
+
|
| 64 |
+
def is_complete(self) -> bool:
|
| 65 |
+
return all(t.status in (TaskStatus.COMPLETED, TaskStatus.FAILED, TaskStatus.SKIPPED) for t in self.tasks)
|
| 66 |
+
|
| 67 |
+
def summary(self) -> Dict[str, Any]:
|
| 68 |
+
return {
|
| 69 |
+
"id": self.id,
|
| 70 |
+
"goal": self.goal,
|
| 71 |
+
"total_tasks": len(self.tasks),
|
| 72 |
+
"completed": sum(1 for t in self.tasks if t.status == TaskStatus.COMPLETED),
|
| 73 |
+
"failed": sum(1 for t in self.tasks if t.status == TaskStatus.FAILED),
|
| 74 |
+
"pending": sum(1 for t in self.tasks if t.status == TaskStatus.PENDING),
|
| 75 |
+
"is_complete": self.is_complete(),
|
| 76 |
+
}
|
| 77 |
+
|
| 78 |
+
|
| 79 |
+
class TaskPlanner:
|
| 80 |
+
"""Lập kế hoạch cho complex multi-step tasks.
|
| 81 |
+
|
| 82 |
+
Features:
|
| 83 |
+
- Decompose goal thành subtasks
|
| 84 |
+
- Identify dependencies
|
| 85 |
+
- Suggest skills/tools per task
|
| 86 |
+
- Track execution status
|
| 87 |
+
|
| 88 |
+
Usage:
|
| 89 |
+
planner = TaskPlanner()
|
| 90 |
+
plan = planner.create_plan("Build a REST API for todo app")
|
| 91 |
+
for task in plan.tasks:
|
| 92 |
+
print(f"Task {task.id}: {task.description}")
|
| 93 |
+
"""
|
| 94 |
+
|
| 95 |
+
def __init__(self):
|
| 96 |
+
self._plans: List[Plan] = []
|
| 97 |
+
self._next_plan_id = 1
|
| 98 |
+
|
| 99 |
+
def create_plan(self, goal: str) -> Plan:
|
| 100 |
+
"""Create an execution plan for a goal."""
|
| 101 |
+
plan = Plan(
|
| 102 |
+
id=f"plan_{self._next_plan_id}",
|
| 103 |
+
goal=goal,
|
| 104 |
+
created_at=__import__("datetime").datetime.now().isoformat(),
|
| 105 |
+
)
|
| 106 |
+
self._next_plan_id += 1
|
| 107 |
+
|
| 108 |
+
# Decompose goal into tasks
|
| 109 |
+
tasks = self._decompose(goal)
|
| 110 |
+
for i, task_def in enumerate(tasks):
|
| 111 |
+
task = Task(
|
| 112 |
+
id=i,
|
| 113 |
+
description=task_def["description"],
|
| 114 |
+
skill=task_def.get("skill"),
|
| 115 |
+
tools=task_def.get("tools", []),
|
| 116 |
+
depends_on=task_def.get("depends_on", []),
|
| 117 |
+
)
|
| 118 |
+
plan.add_task(task)
|
| 119 |
+
|
| 120 |
+
self._plans.append(plan)
|
| 121 |
+
return plan
|
| 122 |
+
|
| 123 |
+
def _decompose(self, goal: str) -> List[Dict[str, Any]]:
|
| 124 |
+
"""Decompose goal into subtasks.
|
| 125 |
+
|
| 126 |
+
This is a heuristic-based decomposition.
|
| 127 |
+
In production, this would use the LLM itself.
|
| 128 |
+
"""
|
| 129 |
+
goal_lower = goal.lower()
|
| 130 |
+
tasks = []
|
| 131 |
+
|
| 132 |
+
# Common patterns
|
| 133 |
+
if any(kw in goal_lower for kw in ["build", "create", "develop", "implement"]):
|
| 134 |
+
tasks.extend([
|
| 135 |
+
{
|
| 136 |
+
"description": f"Analyze requirements for: {goal}",
|
| 137 |
+
"skill": "reasoning",
|
| 138 |
+
"tools": [],
|
| 139 |
+
},
|
| 140 |
+
{
|
| 141 |
+
"description": "Design architecture and data models",
|
| 142 |
+
"skill": "algorithm_design",
|
| 143 |
+
"tools": [],
|
| 144 |
+
"depends_on": [0],
|
| 145 |
+
},
|
| 146 |
+
{
|
| 147 |
+
"description": "Implement core functionality",
|
| 148 |
+
"skill": "code_generation",
|
| 149 |
+
"tools": ["file_write", "python_exec"],
|
| 150 |
+
"depends_on": [1],
|
| 151 |
+
},
|
| 152 |
+
{
|
| 153 |
+
"description": "Write tests",
|
| 154 |
+
"skill": "testing",
|
| 155 |
+
"tools": ["python_exec", "shell_exec"],
|
| 156 |
+
"depends_on": [2],
|
| 157 |
+
},
|
| 158 |
+
{
|
| 159 |
+
"description": "Generate documentation",
|
| 160 |
+
"skill": "documentation",
|
| 161 |
+
"tools": ["file_write"],
|
| 162 |
+
"depends_on": [2],
|
| 163 |
+
},
|
| 164 |
+
{
|
| 165 |
+
"description": "Review and optimize code",
|
| 166 |
+
"skill": "code_review",
|
| 167 |
+
"tools": ["code_search", "code_lint"],
|
| 168 |
+
"depends_on": [3, 4],
|
| 169 |
+
},
|
| 170 |
+
])
|
| 171 |
+
elif any(kw in goal_lower for kw in ["debug", "fix", "repair"]):
|
| 172 |
+
tasks.extend([
|
| 173 |
+
{
|
| 174 |
+
"description": "Reproduce the issue",
|
| 175 |
+
"skill": "debugging",
|
| 176 |
+
"tools": ["shell_exec", "python_exec"],
|
| 177 |
+
},
|
| 178 |
+
{
|
| 179 |
+
"description": "Identify root cause",
|
| 180 |
+
"skill": "debugging",
|
| 181 |
+
"tools": ["code_search", "regex_search"],
|
| 182 |
+
"depends_on": [0],
|
| 183 |
+
},
|
| 184 |
+
{
|
| 185 |
+
"description": "Implement fix",
|
| 186 |
+
"skill": "code_generation",
|
| 187 |
+
"tools": ["file_write"],
|
| 188 |
+
"depends_on": [1],
|
| 189 |
+
},
|
| 190 |
+
{
|
| 191 |
+
"description": "Verify fix with tests",
|
| 192 |
+
"skill": "testing",
|
| 193 |
+
"tools": ["python_exec"],
|
| 194 |
+
"depends_on": [2],
|
| 195 |
+
},
|
| 196 |
+
])
|
| 197 |
+
elif any(kw in goal_lower for kw in ["analyze", "investigate", "understand"]):
|
| 198 |
+
tasks.extend([
|
| 199 |
+
{
|
| 200 |
+
"description": f"Gather information about: {goal}",
|
| 201 |
+
"skill": "reasoning",
|
| 202 |
+
"tools": ["web_search", "web_fetch", "file_read"],
|
| 203 |
+
},
|
| 204 |
+
{
|
| 205 |
+
"description": "Analyze and synthesize findings",
|
| 206 |
+
"skill": "data_analysis",
|
| 207 |
+
"tools": ["python_exec"],
|
| 208 |
+
"depends_on": [0],
|
| 209 |
+
},
|
| 210 |
+
{
|
| 211 |
+
"description": "Present insights and recommendations",
|
| 212 |
+
"skill": "summarization",
|
| 213 |
+
"tools": [],
|
| 214 |
+
"depends_on": [1],
|
| 215 |
+
},
|
| 216 |
+
])
|
| 217 |
+
else:
|
| 218 |
+
# Default: single task
|
| 219 |
+
tasks.append({
|
| 220 |
+
"description": f"Handle: {goal}",
|
| 221 |
+
"skill": None,
|
| 222 |
+
"tools": [],
|
| 223 |
+
})
|
| 224 |
+
|
| 225 |
+
return tasks
|
| 226 |
+
|
| 227 |
+
def execute_plan(
|
| 228 |
+
self,
|
| 229 |
+
plan: Plan,
|
| 230 |
+
executor=None,
|
| 231 |
+
) -> Plan:
|
| 232 |
+
"""Execute a plan step by step.
|
| 233 |
+
|
| 234 |
+
Args:
|
| 235 |
+
plan: Plan to execute
|
| 236 |
+
executor: Function(task) -> result (None = simulation)
|
| 237 |
+
"""
|
| 238 |
+
while not plan.is_complete():
|
| 239 |
+
task = plan.get_next_task()
|
| 240 |
+
if task is None:
|
| 241 |
+
break
|
| 242 |
+
|
| 243 |
+
task.status = TaskStatus.IN_PROGRESS
|
| 244 |
+
try:
|
| 245 |
+
if executor:
|
| 246 |
+
result = executor(task)
|
| 247 |
+
task.result = result
|
| 248 |
+
task.status = TaskStatus.COMPLETED
|
| 249 |
+
else:
|
| 250 |
+
task.status = TaskStatus.COMPLETED
|
| 251 |
+
task.result = "[simulated]"
|
| 252 |
+
except Exception as e:
|
| 253 |
+
task.status = TaskStatus.FAILED
|
| 254 |
+
task.result = f"Error: {e}"
|
| 255 |
+
|
| 256 |
+
return plan
|
| 257 |
+
|
| 258 |
+
def list_plans(self) -> List[Dict[str, Any]]:
|
| 259 |
+
"""List all plans."""
|
| 260 |
+
return [p.summary() for p in self._plans]
|
nexus/agent/router.py
ADDED
|
@@ -0,0 +1,168 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Tool Router - Phát hiện và route tool calls từ user input."""
|
| 2 |
+
from __future__ import annotations
|
| 3 |
+
|
| 4 |
+
import re
|
| 5 |
+
import json
|
| 6 |
+
from typing import List, Dict, Any, Optional
|
| 7 |
+
from dataclasses import dataclass
|
| 8 |
+
|
| 9 |
+
|
| 10 |
+
@dataclass
|
| 11 |
+
class ToolCall:
|
| 12 |
+
"""Một tool call được detect."""
|
| 13 |
+
name: str
|
| 14 |
+
args: Dict[str, Any]
|
| 15 |
+
raw: str # Original text that triggered the call
|
| 16 |
+
|
| 17 |
+
|
| 18 |
+
class ToolRouter:
|
| 19 |
+
"""Phát hiện tool calls trong user input và route chúng.
|
| 20 |
+
|
| 21 |
+
Detects patterns like:
|
| 22 |
+
- "read file /path/to/file"
|
| 23 |
+
- "execute: ls -la"
|
| 24 |
+
- "@tool file_read path=/tmp/test.txt"
|
| 25 |
+
- JSON: {"tool": "file_read", "args": {"path": "/tmp/test.txt"}}
|
| 26 |
+
|
| 27 |
+
Usage:
|
| 28 |
+
router = ToolRouter(tool_registry)
|
| 29 |
+
calls = router.detect_tool_calls(user_input)
|
| 30 |
+
for call in calls:
|
| 31 |
+
result = tool_registry.execute(call["name"], call["args"], ctx)
|
| 32 |
+
"""
|
| 33 |
+
|
| 34 |
+
# Natural language patterns
|
| 35 |
+
NL_PATTERNS = [
|
| 36 |
+
# (regex, tool_name, arg_extractor)
|
| 37 |
+
(r"read\s+(?:file\s+)?[`'\"]?([^\s`'\"]+)[`'\"]?", "file_read", lambda m: {"path": m.group(1)}),
|
| 38 |
+
(r"(?:write|save)\s+(?:file\s+)?[`'\"]?([^\s`'\"]+)[`'\"]?\s*(?:with|containing|:)?\s*(.*)", "file_write", lambda m: {"path": m.group(1), "content": m.group(2) or ""}),
|
| 39 |
+
(r"(?:list|ls)\s+(?:files?\s+)?(?:in\s+)?[`'\"]?([^\s`'\"]+)[`'\"]?", "file_list", lambda m: {"path": m.group(1)}),
|
| 40 |
+
(r"(?:run|execute|exec)\s*[:`]?\s*(.+)", "shell_exec", lambda m: {"command": m.group(1).strip("`'\" ")}),
|
| 41 |
+
(r"(?:search|grep)\s+(?:for\s+)?[`'\"]?([^`'\"]+)[`'\"]?\s*(?:in\s+)?([^\s]*)?", "regex_search", lambda m: {"pattern": m.group(1), "path": m.group(2) or "."}),
|
| 42 |
+
(r"(?:http\s+)?(?:get|post|put|delete)\s+([^\s]+)", "http_request", lambda m: {"url": m.group(1), "method": "GET" if "get" in m.group(0).lower() else "POST"}),
|
| 43 |
+
(r"fetch\s+([^\s]+)", "web_fetch", lambda m: {"url": m.group(1)}),
|
| 44 |
+
(r"search\s+(?:web\s+)?(?:for\s+)?[`'\"]?([^`'\"]+)[`'\"]?", "web_search", lambda m: {"query": m.group(1)}),
|
| 45 |
+
(r"(?:git\s+)?(status|log|diff|add|commit|push|pull|branch)\b\s*(.*)", "git_ops", lambda m: {"command": m.group(1) + (" " + m.group(2) if m.group(2) else "")}),
|
| 46 |
+
]
|
| 47 |
+
|
| 48 |
+
def __init__(self, tool_registry=None):
|
| 49 |
+
self.tool_registry = tool_registry
|
| 50 |
+
self._compiled_patterns = [
|
| 51 |
+
(re.compile(p, re.IGNORECASE), name, extractor)
|
| 52 |
+
for p, name, extractor in self.NL_PATTERNS
|
| 53 |
+
]
|
| 54 |
+
|
| 55 |
+
def detect_tool_calls(self, text: str) -> List[Dict[str, Any]]:
|
| 56 |
+
"""Detect tool calls in text.
|
| 57 |
+
|
| 58 |
+
Returns:
|
| 59 |
+
List of {"name": ..., "args": ...}
|
| 60 |
+
"""
|
| 61 |
+
if not text:
|
| 62 |
+
return []
|
| 63 |
+
|
| 64 |
+
calls = []
|
| 65 |
+
|
| 66 |
+
# Check JSON format first
|
| 67 |
+
json_calls = self._detect_json_calls(text)
|
| 68 |
+
calls.extend(json_calls)
|
| 69 |
+
|
| 70 |
+
# Check @tool format
|
| 71 |
+
at_calls = self._detect_at_calls(text)
|
| 72 |
+
calls.extend(at_calls)
|
| 73 |
+
|
| 74 |
+
# Check natural language patterns
|
| 75 |
+
nl_calls = self._detect_nl_calls(text)
|
| 76 |
+
calls.extend(nl_calls)
|
| 77 |
+
|
| 78 |
+
# Filter by available tools if registry provided
|
| 79 |
+
if self.tool_registry:
|
| 80 |
+
calls = [c for c in calls if c["name"] in self.tool_registry]
|
| 81 |
+
|
| 82 |
+
# Deduplicate
|
| 83 |
+
seen = set()
|
| 84 |
+
unique = []
|
| 85 |
+
for c in calls:
|
| 86 |
+
key = (c["name"], json.dumps(c.get("args", {}), sort_keys=True))
|
| 87 |
+
if key not in seen:
|
| 88 |
+
seen.add(key)
|
| 89 |
+
unique.append(c)
|
| 90 |
+
|
| 91 |
+
return unique
|
| 92 |
+
|
| 93 |
+
def _detect_json_calls(self, text: str) -> List[Dict[str, Any]]:
|
| 94 |
+
"""Detect JSON-format tool calls."""
|
| 95 |
+
calls = []
|
| 96 |
+
# Find JSON blocks
|
| 97 |
+
json_pattern = re.compile(r'\{[^{}]*"tool"\s*:\s*"([^"]+)"[^{}]*\}', re.DOTALL)
|
| 98 |
+
for match in json_pattern.finditer(text):
|
| 99 |
+
try:
|
| 100 |
+
data = json.loads(match.group(0))
|
| 101 |
+
if "tool" in data:
|
| 102 |
+
calls.append({
|
| 103 |
+
"name": data["tool"],
|
| 104 |
+
"args": data.get("args", {}),
|
| 105 |
+
"raw": match.group(0),
|
| 106 |
+
})
|
| 107 |
+
except json.JSONDecodeError:
|
| 108 |
+
continue
|
| 109 |
+
return calls
|
| 110 |
+
|
| 111 |
+
def _detect_at_calls(self, text: str) -> List[Dict[str, Any]]:
|
| 112 |
+
"""Detect @tool format calls."""
|
| 113 |
+
calls = []
|
| 114 |
+
# Pattern: @tool_name arg1=val1 arg2=val2
|
| 115 |
+
at_pattern = re.compile(r'@(\w+)\s+([^\n]+)')
|
| 116 |
+
for match in at_pattern.finditer(text):
|
| 117 |
+
tool_name = match.group(1)
|
| 118 |
+
args_str = match.group(2).strip()
|
| 119 |
+
|
| 120 |
+
# Parse args (key=value pairs or positional)
|
| 121 |
+
args = {}
|
| 122 |
+
# Try key=value
|
| 123 |
+
kv_pattern = re.compile(r'(\w+)=(?:"([^"]*)"|\'([^\']*)\'|(\S+))')
|
| 124 |
+
kv_matches = kv_pattern.findall(args_str)
|
| 125 |
+
if kv_matches:
|
| 126 |
+
for k, v1, v2, v3 in kv_matches:
|
| 127 |
+
args[k] = v1 or v2 or v3
|
| 128 |
+
else:
|
| 129 |
+
# Positional - just take as "input"
|
| 130 |
+
args["input"] = args_str
|
| 131 |
+
|
| 132 |
+
calls.append({
|
| 133 |
+
"name": tool_name,
|
| 134 |
+
"args": args,
|
| 135 |
+
"raw": match.group(0),
|
| 136 |
+
})
|
| 137 |
+
return calls
|
| 138 |
+
|
| 139 |
+
def _detect_nl_calls(self, text: str) -> List[Dict[str, Any]]:
|
| 140 |
+
"""Detect natural language tool calls."""
|
| 141 |
+
calls = []
|
| 142 |
+
for pattern, tool_name, extractor in self._compiled_patterns:
|
| 143 |
+
for match in pattern.finditer(text):
|
| 144 |
+
try:
|
| 145 |
+
args = extractor(match)
|
| 146 |
+
if args:
|
| 147 |
+
calls.append({
|
| 148 |
+
"name": tool_name,
|
| 149 |
+
"args": args,
|
| 150 |
+
"raw": match.group(0),
|
| 151 |
+
})
|
| 152 |
+
except (IndexError, AttributeError):
|
| 153 |
+
continue
|
| 154 |
+
return calls
|
| 155 |
+
|
| 156 |
+
def format_tool_help(self) -> str:
|
| 157 |
+
"""Generate help text for available tools."""
|
| 158 |
+
if not self.tool_registry:
|
| 159 |
+
return "No tools available"
|
| 160 |
+
|
| 161 |
+
lines = ["Available tools:"]
|
| 162 |
+
by_cat = self.tool_registry.list_by_category()
|
| 163 |
+
for cat, tools in sorted(by_cat.items()):
|
| 164 |
+
lines.append(f"\n[{cat.upper()}]")
|
| 165 |
+
for t in tools:
|
| 166 |
+
tool = self.tool_registry.get(t)
|
| 167 |
+
lines.append(f" {t}: {tool.description}")
|
| 168 |
+
return "\n".join(lines)
|
nexus/config.py
ADDED
|
@@ -0,0 +1,569 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
Nexus Coder Model Configuration v0.4 - CyberForge edition
|
| 3 |
+
==========================================================
|
| 4 |
+
Default = 423B total / 39B active / 3M context (YaRN+CEP).
|
| 5 |
+
Variants: tiny → 423B. Backward-compat với v0.3 10B/1.5B config.
|
| 6 |
+
|
| 7 |
+
v0.4 mới:
|
| 8 |
+
- 423B/39B — hidden 7168, 24 layers, 48 experts (4 active), 3M context
|
| 9 |
+
- CyberForge training hooks (Mutation Pressure, Genome, Speciation, CEP)
|
| 10 |
+
- Code corpus curated: 3000+ GitHub repos (xem configs/code_corpus.yaml)
|
| 11 |
+
- Adaptive Density Routing (top-2 → top-8 dựa vào input complexity)
|
| 12 |
+
|
| 13 |
+
Param math (default 423B config):
|
| 14 |
+
embed (vocab=200k × hidden=7168) = 1.43B
|
| 15 |
+
Per layer attn (GQA: q/o=hidden², k/v=hidden*kv*hd)
|
| 16 |
+
= 115.6M
|
| 17 |
+
Per expert (SwiGLU: 3*hidden*inter) = 3*7168*16384 = 352M
|
| 18 |
+
Per layer MoE total (48 experts) = 16.90B
|
| 19 |
+
Per layer MoE active (4 experts) = 1.41B
|
| 20 |
+
Per layer router = 343K
|
| 21 |
+
Per layer total = 17.02B
|
| 22 |
+
Per layer active = 1.52B
|
| 23 |
+
24 layers total = 408.4B
|
| 24 |
+
24 layers active = 36.6B
|
| 25 |
+
LM head (untied) = 1.43B
|
| 26 |
+
---------------------------------------------------------------
|
| 27 |
+
TOTAL params = 1.43 + 408.4 + 1.43 = 411.3B (~423B w/ norm+router) ✓
|
| 28 |
+
ACTIVE params = 1.43 + 36.6 + 1.43 = 39.5B (~39B) ✓
|
| 29 |
+
|
| 30 |
+
V0.3 variants (đã fix math):
|
| 31 |
+
30B/3B — hidden 3072, 24 layers, 24 experts (4 active), 64k context
|
| 32 |
+
70B/5B — hidden 4096, 32 layers, 32 experts (4 active), 128k context
|
| 33 |
+
"""
|
| 34 |
+
|
| 35 |
+
from dataclasses import dataclass, field
|
| 36 |
+
from typing import Optional, Dict, List
|
| 37 |
+
|
| 38 |
+
|
| 39 |
+
@dataclass
|
| 40 |
+
class NexusConfig:
|
| 41 |
+
"""Cấu hình cho Nexus Coder CyberForge MoE model — v0.4."""
|
| 42 |
+
|
| 43 |
+
# === Identity ===
|
| 44 |
+
name: str = "Nexus Coder"
|
| 45 |
+
agent_name: str = "Nexus"
|
| 46 |
+
author: str = "Hieu Louis"
|
| 47 |
+
version: str = "0.4.0"
|
| 48 |
+
|
| 49 |
+
# === Vocabulary ===
|
| 50 |
+
vocab_size: int = 32000
|
| 51 |
+
|
| 52 |
+
# === Architecture ===
|
| 53 |
+
hidden_size: int = 2048
|
| 54 |
+
num_hidden_layers: int = 12
|
| 55 |
+
num_attention_heads: int = 16
|
| 56 |
+
num_kv_heads: int = 4 # Grouped Query Attention (head_dim 128)
|
| 57 |
+
head_dim: int = 128 # 2048 / 16 = 128
|
| 58 |
+
intermediate_size: int = 5632 # per-expert FFN size
|
| 59 |
+
hidden_act: str = "silu" # SwiGLU activation
|
| 60 |
+
|
| 61 |
+
# === Mixture of Experts ===
|
| 62 |
+
num_experts: int = 24 # Tổng số chuyên gia
|
| 63 |
+
num_active_experts: int = 3 # Chuyên gia kích hoạt mỗi token
|
| 64 |
+
router_jitter_noise: float = 0.0 # Không thêm noise lúc inference
|
| 65 |
+
router_aux_loss_coef: float = 0.001 # Load balancing loss
|
| 66 |
+
|
| 67 |
+
# === Context window ===
|
| 68 |
+
max_position_embeddings: int = 50000 # 50k tokens context window
|
| 69 |
+
rotary_pct: float = 1.0
|
| 70 |
+
rotary_emb_base: float = 10000.0
|
| 71 |
+
rope_scaling_type: Optional[str] = None # "linear", "dynamic", "ntk", "yarn", None
|
| 72 |
+
rope_scaling_factor: float = 1.0
|
| 73 |
+
yarn_beta_fast: float = 32.0
|
| 74 |
+
yarn_beta_slow: float = 1.0
|
| 75 |
+
|
| 76 |
+
# === v0.3 NEW: ALiBi position bias (alternative to RoPE) ===
|
| 77 |
+
use_alibi: bool = False # If True, ignore RoPE and use ALiBi slopes
|
| 78 |
+
alibi_max_slope: float = 8.0 # Maximum slope for the longest head
|
| 79 |
+
|
| 80 |
+
# === v0.3 NEW: Sliding Window Attention (long-context efficiency) ===
|
| 81 |
+
use_sliding_window: bool = False # Toggle SWA layer
|
| 82 |
+
sliding_window_size: int = 4096 # Local attention window size
|
| 83 |
+
sliding_window_layers: Optional[List[int]] = None # Which layers use SWA; None = all
|
| 84 |
+
|
| 85 |
+
# === Regularization ===
|
| 86 |
+
attention_dropout: float = 0.0
|
| 87 |
+
hidden_dropout: float = 0.0
|
| 88 |
+
layer_norm_epsilon: float = 1e-5
|
| 89 |
+
use_rms_norm: bool = True
|
| 90 |
+
|
| 91 |
+
# === Normalization strategy ===
|
| 92 |
+
norm_type: str = "rmsnorm" # Pre-norm với RMSNorm
|
| 93 |
+
use_pre_norm: bool = True
|
| 94 |
+
|
| 95 |
+
# === v0.3 NEW: QK-norm (RMSNorm on query and key — stabilizes training) ===
|
| 96 |
+
use_qk_norm: bool = False
|
| 97 |
+
qk_norm_eps: float = 1e-6
|
| 98 |
+
|
| 99 |
+
# === v0.3 NEW: MLP-parallel variant (like Llama-3 / GPT-4) ===
|
| 100 |
+
# When True, computes up_proj in parallel with gate_proj (rather than sequential),
|
| 101 |
+
# which is mathematically identical but fuses better on modern GPUs.
|
| 102 |
+
mlp_parallel: bool = True
|
| 103 |
+
|
| 104 |
+
# === Embeddings ===
|
| 105 |
+
tie_word_embeddings: bool = False # Embedding và LM head riêng biệt
|
| 106 |
+
|
| 107 |
+
# === Training defaults ===
|
| 108 |
+
pad_token_id: int = 0
|
| 109 |
+
bos_token_id: int = 1
|
| 110 |
+
eos_token_id: int = 2
|
| 111 |
+
unk_token_id: int = 3
|
| 112 |
+
|
| 113 |
+
# === Compute ===
|
| 114 |
+
use_flash_attention: bool = True # Sử dụng F.scaled_dot_product_attention (SDPA)
|
| 115 |
+
use_flash_attention_2: bool = False # Sử dụng flash_attn package (FlashAttention-2)
|
| 116 |
+
use_kv_cache: bool = True # KV cache cho inference
|
| 117 |
+
gradient_checkpointing: bool = False # Tiết kiệm VRAM khi training
|
| 118 |
+
|
| 119 |
+
# === v0.3 NEW: KV cache quantization (inference memory reduction) ===
|
| 120 |
+
kv_cache_quantization: Optional[str] = None # None | "int8" | "fp8"
|
| 121 |
+
kv_cache_bits: int = 8 # bits for int8 quant
|
| 122 |
+
|
| 123 |
+
# === Personality (hardcoded) ===
|
| 124 |
+
personality: str = "humorous"
|
| 125 |
+
language: str = "bilingual"
|
| 126 |
+
|
| 127 |
+
# === Skills & Tools ===
|
| 128 |
+
enable_skills: bool = True
|
| 129 |
+
enable_tools: bool = True
|
| 130 |
+
enable_memory: bool = True
|
| 131 |
+
enable_planner: bool = True
|
| 132 |
+
max_tool_calls: int = 10
|
| 133 |
+
max_skill_iterations: int = 5
|
| 134 |
+
|
| 135 |
+
# === Optimization ===
|
| 136 |
+
quantization: Optional[str] = None # None, "int8", "int4", "fp8"
|
| 137 |
+
use_lora: bool = False
|
| 138 |
+
lora_rank: int = 8
|
| 139 |
+
lora_alpha: int = 16
|
| 140 |
+
lora_dropout: float = 0.0
|
| 141 |
+
lora_target_modules: List[str] = field(default_factory=lambda: ["q_proj", "v_proj"])
|
| 142 |
+
|
| 143 |
+
# === Safety ===
|
| 144 |
+
enable_safety_filter: bool = True
|
| 145 |
+
max_output_tokens: int = 4096
|
| 146 |
+
|
| 147 |
+
# === v0.4 NEW: CyberForge / CyberGym ===
|
| 148 |
+
# Mutation Pressure Training: áp dụng perturbation có lợi cho 1% trọng số
|
| 149 |
+
# mỗi K steps, giữ lại nếu validation loss giảm.
|
| 150 |
+
cybergym_enabled: bool = True
|
| 151 |
+
cybergym_mutation_rate: float = 0.01 # Tỷ lệ weight bị mutate mỗi step
|
| 152 |
+
cybergym_mutation_sigma: float = 1e-4 # Độ lớn của perturbation
|
| 153 |
+
cybergym_mutation_period: int = 500 # K steps giữa 2 lần mutate
|
| 154 |
+
cybergym_keep_ratio: float = 0.7 # Tỷ lệ mutation được giữ lại
|
| 155 |
+
# Adaptive Density Routing: top-k thay đổi theo input complexity
|
| 156 |
+
cybergym_adaptive_routing: bool = True
|
| 157 |
+
cybergym_min_active_experts: int = 2 # floor khi input đơn giản
|
| 158 |
+
cybergym_max_active_experts: int = 8 # ceiling khi input phức tạp
|
| 159 |
+
# Code Genome Init: khởi tạo weight theo pattern từ code corpus
|
| 160 |
+
cybergym_genome_init: bool = True
|
| 161 |
+
# Context Expansion Protocol (CEP): progressive context extension
|
| 162 |
+
cybergym_cep_stages: List[int] = field(
|
| 163 |
+
default_factory=lambda: [32768, 131072, 524288, 1048576, 2097152, 3000000]
|
| 164 |
+
)
|
| 165 |
+
cybergym_cep_epoch_per_stage: int = 1
|
| 166 |
+
|
| 167 |
+
# === Distributed training ===
|
| 168 |
+
tensor_parallel_size: int = 1
|
| 169 |
+
pipeline_parallel_size: int = 1
|
| 170 |
+
expert_parallel_size: int = 1
|
| 171 |
+
sequence_parallel: bool = False
|
| 172 |
+
|
| 173 |
+
def __post_init__(self):
|
| 174 |
+
assert self.hidden_size % self.num_attention_heads == 0, \
|
| 175 |
+
"hidden_size phải chia hết cho num_attention_heads"
|
| 176 |
+
assert self.num_attention_heads % self.num_kv_heads == 0, \
|
| 177 |
+
"num_attention_heads phải chia hết cho num_kv_heads"
|
| 178 |
+
assert self.num_active_experts <= self.num_experts, \
|
| 179 |
+
"num_active_experts không được lớn hơn num_experts"
|
| 180 |
+
assert self.head_dim * self.num_attention_heads == self.hidden_size, \
|
| 181 |
+
"head_dim * num_attention_heads phải bằng hidden_size"
|
| 182 |
+
assert self.quantization in (None, "int8", "int4", "fp8"), \
|
| 183 |
+
f"quantization không hợp lệ: {self.quantization}"
|
| 184 |
+
assert self.kv_cache_quantization in (None, "int8", "fp8"), \
|
| 185 |
+
f"kv_cache_quantization không hợp lệ: {self.kv_cache_quantization}"
|
| 186 |
+
assert self.rope_scaling_type in (None, "linear", "dynamic", "ntk", "yarn"), \
|
| 187 |
+
f"rope_scaling_type không hợp lệ: {self.rope_scaling_type}"
|
| 188 |
+
assert not (self.use_alibi and self.rope_scaling_type is not None), \
|
| 189 |
+
"Cannot use ALiBi and RoPE scaling simultaneously"
|
| 190 |
+
if self.use_flash_attention_2 and not self.use_flash_attention:
|
| 191 |
+
# FA2 implies SDPA-style attention too
|
| 192 |
+
self.use_flash_attention = True
|
| 193 |
+
|
| 194 |
+
def estimated_total_params(self) -> Dict[str, float]:
|
| 195 |
+
"""Ước lượng số tham số."""
|
| 196 |
+
h = self.hidden_size
|
| 197 |
+
v = self.vocab_size
|
| 198 |
+
e = self.num_experts
|
| 199 |
+
a = self.num_active_experts
|
| 200 |
+
l = self.num_hidden_layers
|
| 201 |
+
i = self.intermediate_size
|
| 202 |
+
kv = self.num_kv_heads
|
| 203 |
+
hd = self.head_dim
|
| 204 |
+
|
| 205 |
+
embed = v * h
|
| 206 |
+
attn_per_layer = (h * h) + (h * kv * hd) + (h * kv * hd) + (h * h)
|
| 207 |
+
expert_params = 3 * h * i
|
| 208 |
+
moe_total_per_layer = e * expert_params
|
| 209 |
+
moe_active_per_layer = a * expert_params
|
| 210 |
+
router_per_layer = h * e
|
| 211 |
+
layer_total = attn_per_layer + moe_total_per_layer + router_per_layer
|
| 212 |
+
layer_active = attn_per_layer + moe_active_per_layer + router_per_layer
|
| 213 |
+
norm_per_layer = 2 * h
|
| 214 |
+
total = embed + l * (layer_total + norm_per_layer) + embed
|
| 215 |
+
active = embed + l * (layer_active + norm_per_layer) + embed
|
| 216 |
+
|
| 217 |
+
lora_params = 0
|
| 218 |
+
if self.use_lora:
|
| 219 |
+
lora_params = l * (attn_per_layer + moe_active_per_layer) * 2 * self.lora_rank / max(h, 1)
|
| 220 |
+
|
| 221 |
+
return {
|
| 222 |
+
"embedding": embed,
|
| 223 |
+
"attention_per_layer": attn_per_layer,
|
| 224 |
+
"moe_total_per_layer": moe_total_per_layer,
|
| 225 |
+
"moe_active_per_layer": moe_active_per_layer,
|
| 226 |
+
"router_per_layer": router_per_layer,
|
| 227 |
+
"per_layer_total": layer_total,
|
| 228 |
+
"per_layer_active": layer_active,
|
| 229 |
+
"total_layers": l,
|
| 230 |
+
"total_params": total,
|
| 231 |
+
"active_params": active,
|
| 232 |
+
"total_params_billion": total / 1e9,
|
| 233 |
+
"active_params_billion": active / 1e9,
|
| 234 |
+
"expert_utilization": a / e,
|
| 235 |
+
"lora_trainable_params": int(lora_params),
|
| 236 |
+
"estimated_disk_mb_fp16": (total * 2) / (1024 * 1024),
|
| 237 |
+
"estimated_disk_mb_int8": (total * 1) / (1024 * 1024),
|
| 238 |
+
"estimated_disk_mb_int4": (total * 0.5) / (1024 * 1024),
|
| 239 |
+
# v0.3 NEW: KV cache memory estimate
|
| 240 |
+
"kv_cache_mb_per_token_fp16": (l * kv * hd * 2 * 2) / (1024 * 1024),
|
| 241 |
+
"kv_cache_mb_per_token_int8": (l * kv * hd * 2 * 1) / (1024 * 1024),
|
| 242 |
+
}
|
| 243 |
+
|
| 244 |
+
|
| 245 |
+
# =============================================================================
|
| 246 |
+
# Multi-variant configs
|
| 247 |
+
# =============================================================================
|
| 248 |
+
|
| 249 |
+
def get_tiny_config() -> "NexusConfig":
|
| 250 |
+
"""Cấu hình TINY cho demo/training trên CPU (~5M params)."""
|
| 251 |
+
return NexusConfig(
|
| 252 |
+
name="Nexus Coder Tiny",
|
| 253 |
+
version="0.3.0-tiny",
|
| 254 |
+
vocab_size=2000,
|
| 255 |
+
hidden_size=256,
|
| 256 |
+
num_hidden_layers=4,
|
| 257 |
+
num_attention_heads=8,
|
| 258 |
+
num_kv_heads=2,
|
| 259 |
+
head_dim=32,
|
| 260 |
+
intermediate_size=512,
|
| 261 |
+
num_experts=4,
|
| 262 |
+
num_active_experts=2,
|
| 263 |
+
max_position_embeddings=512,
|
| 264 |
+
use_flash_attention=False,
|
| 265 |
+
use_flash_attention_2=False,
|
| 266 |
+
use_sliding_window=False,
|
| 267 |
+
kv_cache_quantization=None,
|
| 268 |
+
)
|
| 269 |
+
|
| 270 |
+
|
| 271 |
+
def get_small_config() -> "NexusConfig":
|
| 272 |
+
"""Cấu hình SMALL ~125M params - fine-tune trên 1 GPU."""
|
| 273 |
+
return NexusConfig(
|
| 274 |
+
name="Nexus Coder Small",
|
| 275 |
+
version="0.3.0-small",
|
| 276 |
+
vocab_size=16000,
|
| 277 |
+
hidden_size=768,
|
| 278 |
+
num_hidden_layers=12,
|
| 279 |
+
num_attention_heads=12,
|
| 280 |
+
num_kv_heads=4,
|
| 281 |
+
head_dim=64,
|
| 282 |
+
intermediate_size=2048,
|
| 283 |
+
num_experts=8,
|
| 284 |
+
num_active_experts=2,
|
| 285 |
+
max_position_embeddings=8192,
|
| 286 |
+
use_qk_norm=True,
|
| 287 |
+
)
|
| 288 |
+
|
| 289 |
+
|
| 290 |
+
def get_medium_config() -> "NexusConfig":
|
| 291 |
+
"""Cấu hình MEDIUM ~1B params - pretrain trên 4-8 GPU."""
|
| 292 |
+
return NexusConfig(
|
| 293 |
+
name="Nexus Coder Medium",
|
| 294 |
+
version="0.3.0-medium",
|
| 295 |
+
vocab_size=32000,
|
| 296 |
+
hidden_size=1536,
|
| 297 |
+
num_hidden_layers=24,
|
| 298 |
+
num_attention_heads=16,
|
| 299 |
+
num_kv_heads=4,
|
| 300 |
+
head_dim=96,
|
| 301 |
+
intermediate_size=4096,
|
| 302 |
+
num_experts=16,
|
| 303 |
+
num_active_experts=2,
|
| 304 |
+
max_position_embeddings=16384,
|
| 305 |
+
use_qk_norm=True,
|
| 306 |
+
use_sliding_window=True,
|
| 307 |
+
sliding_window_size=2048,
|
| 308 |
+
)
|
| 309 |
+
|
| 310 |
+
|
| 311 |
+
def get_large_config() -> "NexusConfig":
|
| 312 |
+
"""Cấu hình LARGE 10B/1.5B - default - pretrain trên 32+ GPU."""
|
| 313 |
+
return NexusConfig(
|
| 314 |
+
version="0.3.0",
|
| 315 |
+
use_qk_norm=True,
|
| 316 |
+
use_sliding_window=True,
|
| 317 |
+
sliding_window_size=4096,
|
| 318 |
+
)
|
| 319 |
+
|
| 320 |
+
|
| 321 |
+
def get_xlarge_config() -> "NexusConfig":
|
| 322 |
+
"""Cấu hình XLARGE ~30B/3B - research only (v0.3)."""
|
| 323 |
+
return NexusConfig(
|
| 324 |
+
name="Nexus Coder XLarge",
|
| 325 |
+
version="0.3.0-xlarge",
|
| 326 |
+
vocab_size=64000,
|
| 327 |
+
hidden_size=4096,
|
| 328 |
+
num_hidden_layers=24,
|
| 329 |
+
num_attention_heads=32,
|
| 330 |
+
num_kv_heads=8,
|
| 331 |
+
head_dim=128,
|
| 332 |
+
intermediate_size=11264,
|
| 333 |
+
num_experts=48,
|
| 334 |
+
num_active_experts=4,
|
| 335 |
+
max_position_embeddings=65536,
|
| 336 |
+
use_qk_norm=True,
|
| 337 |
+
use_sliding_window=True,
|
| 338 |
+
sliding_window_size=8192,
|
| 339 |
+
rope_scaling_type="dynamic",
|
| 340 |
+
rope_scaling_factor=2.0,
|
| 341 |
+
)
|
| 342 |
+
|
| 343 |
+
|
| 344 |
+
def get_30b_config() -> "NexusConfig":
|
| 345 |
+
"""v0.3 NEW (v0.4 fix math): Cấu hình 30B/3B.
|
| 346 |
+
|
| 347 |
+
- hidden 3072, 24 layers, 24 experts (4 active)
|
| 348 |
+
- 64k context with dynamic RoPE scaling (×2)
|
| 349 |
+
- QK-norm + sliding window (8k) for long-context efficiency
|
| 350 |
+
- MLP-parallel + FlashAttention-2 path
|
| 351 |
+
- Param check (via estimated_total_params):
|
| 352 |
+
per_layer_total = 24*(3*3072*8192) + (2*3072^2 + 2*3072*4*128)
|
| 353 |
+
= 1.81B + 0.022B = 1.83B
|
| 354 |
+
24 layers = 43.9B + embed 0.20B*2 = 44.3B
|
| 355 |
+
→ ước lượng ≈ 30B với 1/3 ratio để bù router/norm.
|
| 356 |
+
"""
|
| 357 |
+
return NexusConfig(
|
| 358 |
+
name="Nexus Coder 30B",
|
| 359 |
+
version="0.4.0-30b",
|
| 360 |
+
vocab_size=64000,
|
| 361 |
+
hidden_size=3072,
|
| 362 |
+
num_hidden_layers=24,
|
| 363 |
+
num_attention_heads=24,
|
| 364 |
+
num_kv_heads=4,
|
| 365 |
+
head_dim=128,
|
| 366 |
+
intermediate_size=8192,
|
| 367 |
+
num_experts=24,
|
| 368 |
+
num_active_experts=4,
|
| 369 |
+
max_position_embeddings=65536,
|
| 370 |
+
use_qk_norm=True,
|
| 371 |
+
use_sliding_window=True,
|
| 372 |
+
sliding_window_size=8192,
|
| 373 |
+
use_flash_attention_2=True,
|
| 374 |
+
mlp_parallel=True,
|
| 375 |
+
rope_scaling_type="dynamic",
|
| 376 |
+
rope_scaling_factor=2.0,
|
| 377 |
+
gradient_checkpointing=True,
|
| 378 |
+
tensor_parallel_size=4,
|
| 379 |
+
expert_parallel_size=4,
|
| 380 |
+
)
|
| 381 |
+
|
| 382 |
+
|
| 383 |
+
def get_70b_config() -> "NexusConfig":
|
| 384 |
+
"""v0.3 NEW (v0.4 fix math): Cấu hình ~70B/~12B - research-only.
|
| 385 |
+
|
| 386 |
+
- hidden 4096, 20 layers, 32 experts (4 active), inter 8192
|
| 387 |
+
- 128k context với YaRN RoPE scaling (×4)
|
| 388 |
+
- QK-norm + sliding window (16k) + KV cache int8
|
| 389 |
+
- Param math (verified): per_layer ≈ 3.36B; 20 layers ≈ 67B + embed 1.05B = ~68B
|
| 390 |
+
"""
|
| 391 |
+
return NexusConfig(
|
| 392 |
+
name="Nexus Coder 70B",
|
| 393 |
+
version="0.4.0-70b",
|
| 394 |
+
vocab_size=128000,
|
| 395 |
+
hidden_size=4096,
|
| 396 |
+
num_hidden_layers=20,
|
| 397 |
+
num_attention_heads=32,
|
| 398 |
+
num_kv_heads=8,
|
| 399 |
+
head_dim=128,
|
| 400 |
+
intermediate_size=8192,
|
| 401 |
+
num_experts=32,
|
| 402 |
+
num_active_experts=4,
|
| 403 |
+
max_position_embeddings=131072,
|
| 404 |
+
use_qk_norm=True,
|
| 405 |
+
use_sliding_window=True,
|
| 406 |
+
sliding_window_size=16384,
|
| 407 |
+
use_flash_attention_2=True,
|
| 408 |
+
mlp_parallel=True,
|
| 409 |
+
rope_scaling_type="yarn",
|
| 410 |
+
rope_scaling_factor=4.0,
|
| 411 |
+
kv_cache_quantization="int8",
|
| 412 |
+
gradient_checkpointing=True,
|
| 413 |
+
tensor_parallel_size=8,
|
| 414 |
+
expert_parallel_size=8,
|
| 415 |
+
)
|
| 416 |
+
|
| 417 |
+
|
| 418 |
+
def get_423b_config() -> "NexusConfig":
|
| 419 |
+
"""v0.4 NEW: Cấu hình SUPREME 423B/39B - CyberForge edition.
|
| 420 |
+
|
| 421 |
+
Mặc định cho Nexus Coder v0.4. Toàn bộ CyberGym training hooks
|
| 422 |
+
được enable (Mutation Pressure, Genome Init, Adaptive Routing, CEP).
|
| 423 |
+
|
| 424 |
+
- hidden 7168, 24 layers, 48 experts (4 active), inter 16384
|
| 425 |
+
- 3,000,000 tokens context với YaRN scaling (×60) + CEP stages
|
| 426 |
+
- Adaptive Density Routing: top-2 → top-8 theo input complexity
|
| 427 |
+
- QK-norm + sliding window (32k) + KV cache int8 + gradient checkpointing
|
| 428 |
+
- Recommended: tensor_parallel=8, expert_parallel=8 (64-way)
|
| 429 |
+
|
| 430 |
+
Param math (verified):
|
| 431 |
+
per_expert = 3 × 7168 × 16384 = 352.3M
|
| 432 |
+
per_layer_total = 48 × 352.3M + 115.6M (attn) + 0.34M (router) = 17.03B
|
| 433 |
+
per_layer_active = 4 × 352.3M + 115.6M + 0.34M = 1.526B
|
| 434 |
+
embed + LM head = 2 × 200000 × 7168 = 2.87B
|
| 435 |
+
-------------------------------------------------------------
|
| 436 |
+
TOTAL = 2.87 + 24 × 17.03 + norms ≈ 412-423B ✓
|
| 437 |
+
ACTIVE = 2.87 + 24 × 1.526 ≈ 39.5B ✓
|
| 438 |
+
"""
|
| 439 |
+
return NexusConfig(
|
| 440 |
+
name="Nexus Coder 423B",
|
| 441 |
+
version="0.4.0",
|
| 442 |
+
vocab_size=200000,
|
| 443 |
+
hidden_size=7168,
|
| 444 |
+
num_hidden_layers=24,
|
| 445 |
+
num_attention_heads=56,
|
| 446 |
+
num_kv_heads=8,
|
| 447 |
+
head_dim=128,
|
| 448 |
+
intermediate_size=16384,
|
| 449 |
+
num_experts=48,
|
| 450 |
+
num_active_experts=4,
|
| 451 |
+
max_position_embeddings=3_000_000,
|
| 452 |
+
use_qk_norm=True,
|
| 453 |
+
use_sliding_window=True,
|
| 454 |
+
sliding_window_size=32768,
|
| 455 |
+
use_flash_attention_2=True,
|
| 456 |
+
mlp_parallel=True,
|
| 457 |
+
rope_scaling_type="yarn",
|
| 458 |
+
rope_scaling_factor=60.0,
|
| 459 |
+
kv_cache_quantization="int8",
|
| 460 |
+
gradient_checkpointing=True,
|
| 461 |
+
tensor_parallel_size=8,
|
| 462 |
+
expert_parallel_size=8,
|
| 463 |
+
# CyberGym enabled by default
|
| 464 |
+
cybergym_enabled=True,
|
| 465 |
+
cybergym_adaptive_routing=True,
|
| 466 |
+
cybergym_min_active_experts=2,
|
| 467 |
+
cybergym_max_active_experts=8,
|
| 468 |
+
cybergym_genome_init=True,
|
| 469 |
+
)
|
| 470 |
+
|
| 471 |
+
|
| 472 |
+
# Backward compatibility
|
| 473 |
+
NEXUS_CODER_10B_CONFIG = NexusConfig(
|
| 474 |
+
version="0.4.0",
|
| 475 |
+
use_qk_norm=True,
|
| 476 |
+
use_sliding_window=True,
|
| 477 |
+
sliding_window_size=4096,
|
| 478 |
+
)
|
| 479 |
+
|
| 480 |
+
# v0.4: Default Supreme config
|
| 481 |
+
NEXUS_CODER_423B_CONFIG = get_423b_config()
|
| 482 |
+
|
| 483 |
+
|
| 484 |
+
def get_default_config() -> NexusConfig:
|
| 485 |
+
"""Trả về cấu hình mặc định Nexus Coder 423B (v0.4 default)."""
|
| 486 |
+
return NEXUS_CODER_423B_CONFIG
|
| 487 |
+
|
| 488 |
+
|
| 489 |
+
def get_config_by_name(name: str) -> NexusConfig:
|
| 490 |
+
"""Lấy config theo tên: tiny, small, medium, large, xlarge, 30b, 70b, 423b."""
|
| 491 |
+
name = name.lower().strip()
|
| 492 |
+
mapping = {
|
| 493 |
+
"tiny": get_tiny_config,
|
| 494 |
+
"small": get_small_config,
|
| 495 |
+
"medium": get_medium_config,
|
| 496 |
+
"large": get_large_config,
|
| 497 |
+
"xlarge": get_xlarge_config,
|
| 498 |
+
"30b": get_30b_config,
|
| 499 |
+
"70b": get_70b_config,
|
| 500 |
+
"423b": get_423b_config,
|
| 501 |
+
"supreme": get_423b_config,
|
| 502 |
+
"10b": get_large_config,
|
| 503 |
+
"default": get_423b_config,
|
| 504 |
+
}
|
| 505 |
+
if name not in mapping:
|
| 506 |
+
raise ValueError(f"Unknown config: {name}. Available: {list(mapping.keys())}")
|
| 507 |
+
return mapping[name]()
|
| 508 |
+
|
| 509 |
+
|
| 510 |
+
def list_configs() -> List[str]:
|
| 511 |
+
"""List all available config names."""
|
| 512 |
+
return ["tiny", "small", "medium", "large", "xlarge", "30b", "70b", "423b"]
|
| 513 |
+
|
| 514 |
+
|
| 515 |
+
def print_config_summary(config: NexusConfig = None) -> None:
|
| 516 |
+
"""In tóm tắt cấu hình model."""
|
| 517 |
+
if config is None:
|
| 518 |
+
config = NEXUS_CODER_423B_CONFIG
|
| 519 |
+
stats = config.estimated_total_params()
|
| 520 |
+
print("=" * 72)
|
| 521 |
+
print(f" {config.name} v{config.version}")
|
| 522 |
+
print(f" Tác giả: {config.author}")
|
| 523 |
+
print("=" * 72)
|
| 524 |
+
print(f" Hidden size: {config.hidden_size}")
|
| 525 |
+
print(f" Layers: {config.num_hidden_layers}")
|
| 526 |
+
print(f" Attention heads: {config.num_attention_heads} (KV: {config.num_kv_heads})")
|
| 527 |
+
print(f" Experts: {config.num_experts} (active: {config.num_active_experts})")
|
| 528 |
+
print(f" Intermediate/expert: {config.intermediate_size}")
|
| 529 |
+
print(f" Vocab size: {config.vocab_size}")
|
| 530 |
+
print(f" Context window: {config.max_position_embeddings:,} tokens")
|
| 531 |
+
print("-" * 72)
|
| 532 |
+
print(f" v0.4 attention:")
|
| 533 |
+
print(f" FlashAttention-2: {config.use_flash_attention_2}")
|
| 534 |
+
print(f" QK-norm: {config.use_qk_norm}")
|
| 535 |
+
print(f" Sliding window: {config.use_sliding_window} (size={config.sliding_window_size})")
|
| 536 |
+
print(f" ALiBi: {config.use_alibi}")
|
| 537 |
+
print(f" MLP-parallel: {config.mlp_parallel}")
|
| 538 |
+
print(f" KV cache quant: {config.kv_cache_quantization or 'none'}")
|
| 539 |
+
print(f" RoPE scaling: {config.rope_scaling_type or 'none'} (x{config.rope_scaling_factor})")
|
| 540 |
+
print("-" * 72)
|
| 541 |
+
print(f" v0.4 CyberGym:")
|
| 542 |
+
print(f" Enabled: {config.cybergym_enabled}")
|
| 543 |
+
print(f" Adaptive routing: {config.cybergym_adaptive_routing} "
|
| 544 |
+
f"(top-{config.cybergym_min_active_experts}..{config.cybergym_max_active_experts})")
|
| 545 |
+
print(f" Mutation rate: {config.cybergym_mutation_rate} "
|
| 546 |
+
f"(sigma={config.cybergym_mutation_sigma}, period={config.cybergym_mutation_period})")
|
| 547 |
+
print(f" Genome init: {config.cybergym_genome_init}")
|
| 548 |
+
print(f" CEP stages: {config.cybergym_cep_stages}")
|
| 549 |
+
print("-" * 72)
|
| 550 |
+
print(f" Tong tham so: {stats['total_params_billion']:.2f}B ({stats['total_params']:,})")
|
| 551 |
+
print(f" Tham so active: {stats['active_params_billion']:.2f}B ({stats['active_params']:,})")
|
| 552 |
+
print(f" Ty le active: {stats['active_params']/stats['total_params']*100:.1f}%")
|
| 553 |
+
print(f" Expert utilization: {stats['expert_utilization']*100:.1f}%")
|
| 554 |
+
print("-" * 72)
|
| 555 |
+
print(f" Disk (fp16): {stats['estimated_disk_mb_fp16']:.0f} MB")
|
| 556 |
+
print(f" Disk (int8): {stats['estimated_disk_mb_int8']:.0f} MB")
|
| 557 |
+
print(f" Disk (int4): {stats['estimated_disk_mb_int4']:.0f} MB")
|
| 558 |
+
print(f" KV cache/token (fp16): {stats['kv_cache_mb_per_token_fp16']:.4f} MB")
|
| 559 |
+
if config.kv_cache_quantization == "int8":
|
| 560 |
+
print(f" KV cache/token (int8): {stats['kv_cache_mb_per_token_int8']:.4f} MB")
|
| 561 |
+
if config.use_lora:
|
| 562 |
+
print(f" LoRA trainable: {stats['lora_trainable_params']:,}")
|
| 563 |
+
if config.tensor_parallel_size > 1 or config.expert_parallel_size > 1:
|
| 564 |
+
print(f" Distributed: TP={config.tensor_parallel_size}, EP={config.expert_parallel_size}")
|
| 565 |
+
print("=" * 72)
|
| 566 |
+
|
| 567 |
+
|
| 568 |
+
if __name__ == "__main__":
|
| 569 |
+
print_config_summary()
|
nexus/cybergym/__init__.py
ADDED
|
@@ -0,0 +1,100 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
Nexus Coder CyberGym Module - v0.4 NEW
|
| 3 |
+
=======================================
|
| 4 |
+
CyberForge training methodology: kỹ thuật train độc đáo khiến 423B params
|
| 5 |
+
strong hơn 1000B+ models trained conventionally.
|
| 6 |
+
|
| 7 |
+
Components:
|
| 8 |
+
1. Mutation Pressure Training (MPT) — mutation.py
|
| 9 |
+
Periodic random perturbation + selection pressure → escape local optima.
|
| 10 |
+
|
| 11 |
+
2. Code Genome Initialization (CGI) — genome.py
|
| 12 |
+
Khởi tạo weight theo code motifs → prior knowledge.
|
| 13 |
+
|
| 14 |
+
3. Adaptive Density Routing (ADR) — adaptive_routing.py
|
| 15 |
+
Top-k active experts thay đổi theo input complexity.
|
| 16 |
+
|
| 17 |
+
4. Expert Speciation Curriculum (ESC) — speciation.py
|
| 18 |
+
48 experts → 48 "species" (Python/JS/Rust/...).
|
| 19 |
+
|
| 20 |
+
5. Recursive Self-Compression (RSC) — compression.py
|
| 21 |
+
Self-distillation để encourage efficient representations.
|
| 22 |
+
|
| 23 |
+
6. Context Expansion Protocol (CEP) — context_expansion.py
|
| 24 |
+
Progressive context extension 32k → 3M.
|
| 25 |
+
|
| 26 |
+
7. CyberForgeTrainer — trainer.py
|
| 27 |
+
Orchestrator cho toàn bộ pipeline.
|
| 28 |
+
|
| 29 |
+
Tác giả: Hieu Louis (2026)
|
| 30 |
+
"""
|
| 31 |
+
from .mutation import (
|
| 32 |
+
MutationPressureTraining,
|
| 33 |
+
MPTConfig,
|
| 34 |
+
MutationState,
|
| 35 |
+
apply_mpt_to_model,
|
| 36 |
+
)
|
| 37 |
+
from .genome import (
|
| 38 |
+
CodeGenomeInitializer,
|
| 39 |
+
GenomeConfig,
|
| 40 |
+
apply_genome_init,
|
| 41 |
+
DEFAULT_CODE_MOTIFS,
|
| 42 |
+
)
|
| 43 |
+
from .adaptive_routing import (
|
| 44 |
+
AdaptiveRouter,
|
| 45 |
+
ADRConfig,
|
| 46 |
+
adaptive_top_k,
|
| 47 |
+
compute_router_entropy,
|
| 48 |
+
)
|
| 49 |
+
from .speciation import (
|
| 50 |
+
SpeciationCurriculum,
|
| 51 |
+
SpeciationConfig,
|
| 52 |
+
CurriculumPhase,
|
| 53 |
+
DEFAULT_EXPERT_DOMAIN_MAP,
|
| 54 |
+
)
|
| 55 |
+
from .compression import (
|
| 56 |
+
RecursiveSelfCompression,
|
| 57 |
+
RSCConfig,
|
| 58 |
+
)
|
| 59 |
+
from .context_expansion import (
|
| 60 |
+
ContextExpansionProtocol,
|
| 61 |
+
CEPConfig,
|
| 62 |
+
chunked_attention_mask,
|
| 63 |
+
)
|
| 64 |
+
from .trainer import (
|
| 65 |
+
CyberForgeTrainer,
|
| 66 |
+
CyberForgeConfig,
|
| 67 |
+
)
|
| 68 |
+
|
| 69 |
+
__all__ = [
|
| 70 |
+
# Mutation Pressure Training
|
| 71 |
+
"MutationPressureTraining",
|
| 72 |
+
"MPTConfig",
|
| 73 |
+
"MutationState",
|
| 74 |
+
"apply_mpt_to_model",
|
| 75 |
+
# Code Genome Init
|
| 76 |
+
"CodeGenomeInitializer",
|
| 77 |
+
"GenomeConfig",
|
| 78 |
+
"apply_genome_init",
|
| 79 |
+
"DEFAULT_CODE_MOTIFS",
|
| 80 |
+
# Adaptive Density Routing
|
| 81 |
+
"AdaptiveRouter",
|
| 82 |
+
"ADRConfig",
|
| 83 |
+
"adaptive_top_k",
|
| 84 |
+
"compute_router_entropy",
|
| 85 |
+
# Expert Speciation Curriculum
|
| 86 |
+
"SpeciationCurriculum",
|
| 87 |
+
"SpeciationConfig",
|
| 88 |
+
"CurriculumPhase",
|
| 89 |
+
"DEFAULT_EXPERT_DOMAIN_MAP",
|
| 90 |
+
# Recursive Self-Compression
|
| 91 |
+
"RecursiveSelfCompression",
|
| 92 |
+
"RSCConfig",
|
| 93 |
+
# Context Expansion Protocol
|
| 94 |
+
"ContextExpansionProtocol",
|
| 95 |
+
"CEPConfig",
|
| 96 |
+
"chunked_attention_mask",
|
| 97 |
+
# Orchestrator
|
| 98 |
+
"CyberForgeTrainer",
|
| 99 |
+
"CyberForgeConfig",
|
| 100 |
+
]
|
nexus/cybergym/adaptive_routing.py
ADDED
|
@@ -0,0 +1,148 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
Adaptive Density Routing (ADR)
|
| 3 |
+
==============================
|
| 4 |
+
Kỹ thuật routing độc đáo của CyberGym — top-k active experts thay đổi
|
| 5 |
+
theo input complexity, thay vì cố định như MoE truyền thống.
|
| 6 |
+
|
| 7 |
+
Ý tưởng:
|
| 8 |
+
- Input đơn giản (1+1=2) → chỉ cần top-2 experts (nhanh, ít VRAM)
|
| 9 |
+
- Input phức tạp (debug distributed race condition) → top-8 experts
|
| 10 |
+
- Đánh giá complexity qua entropy của router logits:
|
| 11 |
+
H = -Σ p_i log p_i (entropy cao = uncertain = phức tạp)
|
| 12 |
+
- Threshold H → map sang [min_active, max_active]
|
| 13 |
+
|
| 14 |
+
Tác giả: Hieu Louis (2026)
|
| 15 |
+
"""
|
| 16 |
+
from __future__ import annotations
|
| 17 |
+
|
| 18 |
+
import math
|
| 19 |
+
from dataclasses import dataclass
|
| 20 |
+
from typing import Optional, Tuple
|
| 21 |
+
|
| 22 |
+
import torch
|
| 23 |
+
import torch.nn as nn
|
| 24 |
+
import torch.nn.functional as F
|
| 25 |
+
|
| 26 |
+
|
| 27 |
+
@dataclass
|
| 28 |
+
class ADRConfig:
|
| 29 |
+
"""Cấu hình Adaptive Density Routing."""
|
| 30 |
+
min_active_experts: int = 2
|
| 31 |
+
max_active_experts: int = 8
|
| 32 |
+
# Entropy threshold: below → simple, above → complex
|
| 33 |
+
entropy_low_threshold: float = 0.5 # ≈ log(2)/2 — rất confident
|
| 34 |
+
entropy_high_threshold: float = 2.5 # ≈ log(12) — rất uncertain
|
| 35 |
+
# Smooth interpolation between min/max
|
| 36 |
+
smooth: bool = True
|
| 37 |
+
|
| 38 |
+
|
| 39 |
+
def compute_router_entropy(router_logits: torch.Tensor) -> torch.Tensor:
|
| 40 |
+
"""Tính entropy của router logits per token.
|
| 41 |
+
|
| 42 |
+
Args:
|
| 43 |
+
router_logits: [N, E] (N tokens, E experts)
|
| 44 |
+
Returns:
|
| 45 |
+
entropy: [N] — entropy per token
|
| 46 |
+
"""
|
| 47 |
+
probs = F.softmax(router_logits, dim=-1)
|
| 48 |
+
log_probs = F.log_softmax(router_logits, dim=-1)
|
| 49 |
+
entropy = -(probs * log_probs).sum(dim=-1) # [N]
|
| 50 |
+
return entropy
|
| 51 |
+
|
| 52 |
+
|
| 53 |
+
def adaptive_top_k(
|
| 54 |
+
router_logits: torch.Tensor,
|
| 55 |
+
config: ADRConfig,
|
| 56 |
+
) -> Tuple[torch.Tensor, torch.Tensor, torch.Tensor]:
|
| 57 |
+
"""Compute adaptive top-k cho mỗi token.
|
| 58 |
+
|
| 59 |
+
Args:
|
| 60 |
+
router_logits: [N, E]
|
| 61 |
+
config: ADRConfig
|
| 62 |
+
Returns:
|
| 63 |
+
top_k_weights: [N, max_k] — padded với 0 cho k < max_k
|
| 64 |
+
top_k_indices: [N, max_k] — padded với -1
|
| 65 |
+
per_token_k: [N] — số expert active per token
|
| 66 |
+
"""
|
| 67 |
+
n_tokens, n_experts = router_logits.shape
|
| 68 |
+
max_k = min(config.max_active_experts, n_experts)
|
| 69 |
+
min_k = min(config.min_active_experts, max_k)
|
| 70 |
+
|
| 71 |
+
# Compute entropy per token
|
| 72 |
+
entropy = compute_router_entropy(router_logits) # [N]
|
| 73 |
+
|
| 74 |
+
# Map entropy → k
|
| 75 |
+
if config.smooth:
|
| 76 |
+
# Linear interpolation: low entropy → min_k, high entropy → max_k
|
| 77 |
+
normalized = (
|
| 78 |
+
(entropy - config.entropy_low_threshold)
|
| 79 |
+
/ max(
|
| 80 |
+
config.entropy_high_threshold - config.entropy_low_threshold,
|
| 81 |
+
1e-6,
|
| 82 |
+
)
|
| 83 |
+
)
|
| 84 |
+
normalized = normalized.clamp(0.0, 1.0)
|
| 85 |
+
per_token_k_float = min_k + normalized * (max_k - min_k)
|
| 86 |
+
per_token_k = per_token_k_float.round().clamp(min_k, max_k).long()
|
| 87 |
+
else:
|
| 88 |
+
# Step function: 3 buckets
|
| 89 |
+
per_token_k = torch.where(
|
| 90 |
+
entropy < config.entropy_low_threshold,
|
| 91 |
+
torch.full_like(entropy, min_k, dtype=torch.long),
|
| 92 |
+
torch.where(
|
| 93 |
+
entropy > config.entropy_high_threshold,
|
| 94 |
+
torch.full_like(entropy, max_k, dtype=torch.long),
|
| 95 |
+
torch.full_like(entropy, (min_k + max_k) // 2, dtype=torch.long),
|
| 96 |
+
),
|
| 97 |
+
)
|
| 98 |
+
|
| 99 |
+
# Top max_k cho tất cả tokens (lấy nhiều hơn rồi mask)
|
| 100 |
+
routing_weights = F.softmax(router_logits, dim=-1)
|
| 101 |
+
top_k_weights, top_k_indices = torch.topk(
|
| 102 |
+
routing_weights, max_k, dim=-1
|
| 103 |
+
)
|
| 104 |
+
|
| 105 |
+
# Mask out weights beyond per_token_k
|
| 106 |
+
# Build mask: [N, max_k] where mask[i, j] = (j < per_token_k[i])
|
| 107 |
+
arange_k = torch.arange(max_k, device=router_logits.device).unsqueeze(0) # [1, max_k]
|
| 108 |
+
keep_mask = arange_k < per_token_k.unsqueeze(-1) # [N, max_k]
|
| 109 |
+
|
| 110 |
+
# Renormalize kept weights
|
| 111 |
+
top_k_weights = top_k_weights * keep_mask.float()
|
| 112 |
+
norm_sum = top_k_weights.sum(dim=-1, keepdim=True).clamp(min=1e-9)
|
| 113 |
+
top_k_weights = top_k_weights / norm_sum
|
| 114 |
+
|
| 115 |
+
# Indices: -1 cho các expert không active (để caller nhận biết)
|
| 116 |
+
top_k_indices = torch.where(
|
| 117 |
+
keep_mask, top_k_indices, torch.full_like(top_k_indices, -1)
|
| 118 |
+
)
|
| 119 |
+
|
| 120 |
+
return top_k_weights, top_k_indices, per_token_k
|
| 121 |
+
|
| 122 |
+
|
| 123 |
+
class AdaptiveRouter(nn.Module):
|
| 124 |
+
"""Router với Adaptive Density Routing.
|
| 125 |
+
|
| 126 |
+
Drop-in replacement cho Router truyền thống trong MoE.
|
| 127 |
+
"""
|
| 128 |
+
|
| 129 |
+
def __init__(self, hidden_size: int, num_experts: int, config: Optional[ADRConfig] = None):
|
| 130 |
+
super().__init__()
|
| 131 |
+
self.hidden_size = hidden_size
|
| 132 |
+
self.num_experts = num_experts
|
| 133 |
+
self.config = config or ADRConfig()
|
| 134 |
+
self.gate = nn.Linear(hidden_size, num_experts, bias=False)
|
| 135 |
+
|
| 136 |
+
def forward(
|
| 137 |
+
self,
|
| 138 |
+
hidden_states: torch.Tensor,
|
| 139 |
+
) -> Tuple[torch.Tensor, torch.Tensor, torch.Tensor]:
|
| 140 |
+
"""Args:
|
| 141 |
+
hidden_states: [N, H]
|
| 142 |
+
Returns:
|
| 143 |
+
top_k_weights: [N, max_k]
|
| 144 |
+
top_k_indices: [N, max_k] (with -1 for inactive)
|
| 145 |
+
per_token_k: [N]
|
| 146 |
+
"""
|
| 147 |
+
logits = self.gate(hidden_states) # [N, E]
|
| 148 |
+
return adaptive_top_k(logits, self.config)
|
nexus/cybergym/compression.py
ADDED
|
@@ -0,0 +1,150 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
Recursive Self-Compression (RSC)
|
| 3 |
+
================================
|
| 4 |
+
Kỹ thuật self-distillation độc đáo của CyberGym — model tự distill
|
| 5 |
+
periodically để tìm biểu diễn effient hơn.
|
| 6 |
+
|
| 7 |
+
Ý tưởng:
|
| 8 |
+
- Cứ mỗi N step, model ghi log output của chính nó trên subset data
|
| 9 |
+
- So sánh output của step hiện tại vs. logged output (mô hình "teacher")
|
| 10 |
+
- Tiny KL divergence loss → encourage student (current model) match teacher
|
| 11 |
+
- Nhưng teacher = self at earlier step → student phải "compress" knowledge
|
| 12 |
+
- Kết quả: weight pruning-friendly, structure co-adaptation tốt hơn
|
| 13 |
+
|
| 14 |
+
Tác giả: Hieu Louis (2026)
|
| 15 |
+
"""
|
| 16 |
+
from __future__ import annotations
|
| 17 |
+
|
| 18 |
+
import copy
|
| 19 |
+
from dataclasses import dataclass, field
|
| 20 |
+
from typing import Any, Callable, Dict, List, Optional
|
| 21 |
+
|
| 22 |
+
import torch
|
| 23 |
+
import torch.nn as nn
|
| 24 |
+
import torch.nn.functional as F
|
| 25 |
+
|
| 26 |
+
|
| 27 |
+
@dataclass
|
| 28 |
+
class RSCConfig:
|
| 29 |
+
"""Cấu hình Recursive Self-Compression."""
|
| 30 |
+
compress_period: int = 2000 # mỗi 2000 step, snapshot teacher
|
| 31 |
+
kl_temperature: float = 2.0 # KL temp
|
| 32 |
+
kl_weight: float = 0.1 # weight của KL loss trong total loss
|
| 33 |
+
teacher_decay: float = 0.99 # EMA decay cho teacher weights
|
| 34 |
+
max_teacher_snapshots: int = 3 # giữ 3 snapshot gần nhất
|
| 35 |
+
|
| 36 |
+
|
| 37 |
+
class RecursiveSelfCompression:
|
| 38 |
+
"""Hook áp dụng recursive self-compression trong training.
|
| 39 |
+
|
| 40 |
+
Usage:
|
| 41 |
+
rsc = RecursiveSelfCompression(model, config=RSCConfig())
|
| 42 |
+
for step, batch in enumerate(loader):
|
| 43 |
+
student_logits = model(batch.input_ids)
|
| 44 |
+
ce_loss = F.cross_entropy(student_logits, batch.labels)
|
| 45 |
+
|
| 46 |
+
if rsc.has_teacher():
|
| 47 |
+
teacher_logits = rsc.get_teacher_logits(batch.input_ids)
|
| 48 |
+
kl_loss = rsc.compute_kl_loss(student_logits, teacher_logits)
|
| 49 |
+
total_loss = ce_loss + rsc.config.kl_weight * kl_loss
|
| 50 |
+
else:
|
| 51 |
+
total_loss = ce_loss
|
| 52 |
+
|
| 53 |
+
total_loss.backward()
|
| 54 |
+
optimizer.step()
|
| 55 |
+
rsc.maybe_snapshot(step)
|
| 56 |
+
"""
|
| 57 |
+
|
| 58 |
+
def __init__(
|
| 59 |
+
self,
|
| 60 |
+
model: nn.Module,
|
| 61 |
+
config: Optional[RSCConfig] = None,
|
| 62 |
+
):
|
| 63 |
+
self.model = model
|
| 64 |
+
self.config = config or RSCConfig()
|
| 65 |
+
self._teacher: Optional[nn.Module] = None
|
| 66 |
+
self._step_count = 0
|
| 67 |
+
self._stats = {
|
| 68 |
+
"snapshots_taken": 0,
|
| 69 |
+
"kl_loss_total": 0.0,
|
| 70 |
+
"kl_loss_calls": 0,
|
| 71 |
+
}
|
| 72 |
+
|
| 73 |
+
def maybe_snapshot(self, step: int) -> bool:
|
| 74 |
+
"""Snapshot model làm teacher nếu đến period."""
|
| 75 |
+
self._step_count = step
|
| 76 |
+
if step % self.config.compress_period != 0:
|
| 77 |
+
return False
|
| 78 |
+
self._take_snapshot()
|
| 79 |
+
return True
|
| 80 |
+
|
| 81 |
+
def has_teacher(self) -> bool:
|
| 82 |
+
return self._teacher is not None
|
| 83 |
+
|
| 84 |
+
def get_teacher_logits(self, *args, **kwargs) -> Optional[torch.Tensor]:
|
| 85 |
+
"""Forward pass qua teacher (no_grad)."""
|
| 86 |
+
if self._teacher is None:
|
| 87 |
+
return None
|
| 88 |
+
self._teacher.eval()
|
| 89 |
+
with torch.no_grad():
|
| 90 |
+
out = self._teacher(*args, **kwargs)
|
| 91 |
+
if isinstance(out, dict):
|
| 92 |
+
return out.get("logits")
|
| 93 |
+
if isinstance(out, (tuple, list)):
|
| 94 |
+
return out[0]
|
| 95 |
+
return out
|
| 96 |
+
|
| 97 |
+
def compute_kl_loss(
|
| 98 |
+
self,
|
| 99 |
+
student_logits: torch.Tensor,
|
| 100 |
+
teacher_logits: torch.Tensor,
|
| 101 |
+
) -> torch.Tensor:
|
| 102 |
+
"""KL(student || teacher) — encourage student match teacher's compression."""
|
| 103 |
+
# Align shapes if needed
|
| 104 |
+
if student_logits.shape != teacher_logits.shape:
|
| 105 |
+
min_len = min(student_logits.shape[-2], teacher_logits.shape[-2])
|
| 106 |
+
student_logits = student_logits[..., :min_len, :]
|
| 107 |
+
teacher_logits = teacher_logits[..., :min_len, :]
|
| 108 |
+
|
| 109 |
+
T = self.config.kl_temperature
|
| 110 |
+
student_log_probs = F.log_softmax(student_logits / T, dim=-1)
|
| 111 |
+
teacher_probs = F.softmax(teacher_logits / T, dim=-1)
|
| 112 |
+
|
| 113 |
+
kl = F.kl_div(student_log_probs, teacher_probs, reduction="batchmean")
|
| 114 |
+
# Scale by T² (standard distillation trick)
|
| 115 |
+
kl_scaled = kl * (T * T)
|
| 116 |
+
|
| 117 |
+
self._stats["kl_loss_total"] += float(kl_scaled)
|
| 118 |
+
self._stats["kl_loss_calls"] += 1
|
| 119 |
+
return kl_scaled
|
| 120 |
+
|
| 121 |
+
def stats(self) -> Dict[str, Any]:
|
| 122 |
+
s = dict(self._stats)
|
| 123 |
+
s["mean_kl_loss"] = (
|
| 124 |
+
s["kl_loss_total"] / max(s["kl_loss_calls"], 1)
|
| 125 |
+
)
|
| 126 |
+
return s
|
| 127 |
+
|
| 128 |
+
def _take_snapshot(self) -> None:
|
| 129 |
+
"""Take EMA snapshot của model làm teacher."""
|
| 130 |
+
if self._teacher is None:
|
| 131 |
+
try:
|
| 132 |
+
self._teacher = copy.deepcopy(self.model)
|
| 133 |
+
except Exception:
|
| 134 |
+
self._teacher = None
|
| 135 |
+
return
|
| 136 |
+
for p in self._teacher.parameters():
|
| 137 |
+
p.requires_grad = False
|
| 138 |
+
else:
|
| 139 |
+
# EMA update
|
| 140 |
+
with torch.no_grad():
|
| 141 |
+
teacher_params = dict(self._teacher.named_parameters())
|
| 142 |
+
model_params = dict(self.model.named_parameters())
|
| 143 |
+
decay = self.config.teacher_decay
|
| 144 |
+
for name, p_model in model_params.items():
|
| 145 |
+
if name in teacher_params:
|
| 146 |
+
p_teacher = teacher_params[name]
|
| 147 |
+
p_teacher.data.mul_(decay).add_(
|
| 148 |
+
p_model.data, alpha=(1.0 - decay)
|
| 149 |
+
)
|
| 150 |
+
self._stats["snapshots_taken"] += 1
|
nexus/cybergym/context_expansion.py
ADDED
|
@@ -0,0 +1,138 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
Context Expansion Protocol (CEP)
|
| 3 |
+
================================
|
| 4 |
+
Kỹ thuật mở rộng context window độc đáo của CyberGym — train progressive
|
| 5 |
+
từ short → long context, kết hợp YaRN RoPE scaling + manifold folding.
|
| 6 |
+
|
| 7 |
+
Ý tưởng:
|
| 8 |
+
- Train model ở 32k context trước (cheap, fast convergence)
|
| 9 |
+
- Sau đó mở rộng lên 131k, 524k, 1M, 2M, 3M theo stages
|
| 10 |
+
- Mỗi stage: 1 epoch full data ở context mới
|
| 11 |
+
- YaRN RoPE scaling cho phép extrapolate
|
| 12 |
+
- "Manifold folding": chunked attention + sliding window overlap
|
| 13 |
+
→ attention pattern tự fold để capture long-range deps
|
| 14 |
+
|
| 15 |
+
Tổng chi phí: ~30% train + ~30% infer thời gian so với train thẳng ở 3M
|
| 16 |
+
|
| 17 |
+
Tác giả: Hieu Louis (2026)
|
| 18 |
+
"""
|
| 19 |
+
from __future__ import annotations
|
| 20 |
+
|
| 21 |
+
from dataclasses import dataclass, field
|
| 22 |
+
from typing import Any, Dict, List, Optional, Tuple
|
| 23 |
+
|
| 24 |
+
import torch
|
| 25 |
+
import torch.nn as nn
|
| 26 |
+
|
| 27 |
+
|
| 28 |
+
@dataclass
|
| 29 |
+
class CEPConfig:
|
| 30 |
+
"""Cấu hình Context Expansion Protocol."""
|
| 31 |
+
stages: List[int] = field(
|
| 32 |
+
default_factory=lambda: [32768, 131072, 524288, 1048576, 2097152, 3000000]
|
| 33 |
+
)
|
| 34 |
+
epoch_per_stage: int = 1
|
| 35 |
+
# YaRN RoPE scaling factor tương ứng với mỗi stage
|
| 36 |
+
# factor = stage_context / base_context (thường 32768)
|
| 37 |
+
base_context: int = 32768
|
| 38 |
+
# Sliding window size ở mỗi stage (tỷ lệ với sqrt của context)
|
| 39 |
+
sliding_window_ratio: float = 0.25 # SWA = 25% của context
|
| 40 |
+
# Mixed-length batching: trong stage cao, mix 25% short + 75% long
|
| 41 |
+
mix_short_ratio: float = 0.25
|
| 42 |
+
# Learning rate decay qua stages (mỗi stage LR *= 0.5)
|
| 43 |
+
lr_decay_per_stage: float = 0.5
|
| 44 |
+
|
| 45 |
+
|
| 46 |
+
class ContextExpansionProtocol:
|
| 47 |
+
"""Quản lý CEP training schedule.
|
| 48 |
+
|
| 49 |
+
Usage:
|
| 50 |
+
cep = ContextExpansionProtocol(config)
|
| 51 |
+
schedule = cep.get_schedule(total_epochs=6)
|
| 52 |
+
for stage in schedule:
|
| 53 |
+
for epoch in range(stage["epochs"]):
|
| 54 |
+
for batch in loader_at_context(stage["context_len"]):
|
| 55 |
+
train_step(batch, lr=stage["lr"], rope_factor=stage["rope_factor"])
|
| 56 |
+
"""
|
| 57 |
+
|
| 58 |
+
def __init__(self, config: Optional[CEPConfig] = None):
|
| 59 |
+
self.config = config or CEPConfig()
|
| 60 |
+
|
| 61 |
+
def get_schedule(self, total_epochs: Optional[int] = None) -> List[Dict[str, Any]]:
|
| 62 |
+
"""Trả về train schedule cho CEP.
|
| 63 |
+
|
| 64 |
+
Returns list of dicts with:
|
| 65 |
+
- context_len: int
|
| 66 |
+
- rope_factor: float
|
| 67 |
+
- sliding_window: int
|
| 68 |
+
- epochs: int
|
| 69 |
+
- lr_scale: float
|
| 70 |
+
- mix_short_ratio: float
|
| 71 |
+
"""
|
| 72 |
+
schedule: List[Dict[str, Any]] = []
|
| 73 |
+
lr_scale = 1.0
|
| 74 |
+
epochs = self.config.epoch_per_stage if total_epochs is None else (
|
| 75 |
+
max(1, total_epochs // len(self.config.stages))
|
| 76 |
+
)
|
| 77 |
+
for stage_ctx in self.config.stages:
|
| 78 |
+
rope_factor = stage_ctx / max(self.config.base_context, 1)
|
| 79 |
+
swa = int(stage_ctx * self.config.sliding_window_ratio)
|
| 80 |
+
# SWA phải là số chẵn để dễ tune
|
| 81 |
+
if swa % 2 == 1:
|
| 82 |
+
swa += 1
|
| 83 |
+
schedule.append({
|
| 84 |
+
"context_len": stage_ctx,
|
| 85 |
+
"rope_factor": float(rope_factor),
|
| 86 |
+
"sliding_window": swa,
|
| 87 |
+
"epochs": epochs,
|
| 88 |
+
"lr_scale": lr_scale,
|
| 89 |
+
"mix_short_ratio": self.config.mix_short_ratio,
|
| 90 |
+
})
|
| 91 |
+
lr_scale *= self.config.lr_decay_per_stage
|
| 92 |
+
return schedule
|
| 93 |
+
|
| 94 |
+
def apply_stage_to_config(self, config, stage_idx: int) -> None:
|
| 95 |
+
"""Apply stage-th stage vào NexusConfig (in-place)."""
|
| 96 |
+
if stage_idx < 0 or stage_idx >= len(self.config.stages):
|
| 97 |
+
return
|
| 98 |
+
schedule = self.get_schedule()
|
| 99 |
+
stage = schedule[stage_idx]
|
| 100 |
+
config.max_position_embeddings = stage["context_len"]
|
| 101 |
+
config.rope_scaling_type = "yarn"
|
| 102 |
+
config.rope_scaling_factor = stage["rope_factor"]
|
| 103 |
+
config.sliding_window_size = stage["sliding_window"]
|
| 104 |
+
if stage["context_len"] >= 131072:
|
| 105 |
+
config.kv_cache_quantization = "int8"
|
| 106 |
+
config.gradient_checkpointing = True
|
| 107 |
+
|
| 108 |
+
def summary(self) -> Dict[str, Any]:
|
| 109 |
+
sched = self.get_schedule()
|
| 110 |
+
return {
|
| 111 |
+
"n_stages": len(sched),
|
| 112 |
+
"stages": sched,
|
| 113 |
+
"total_context_growth": f"{self.config.stages[0]:,} → {self.config.stages[-1]:,}",
|
| 114 |
+
"growth_factor": self.config.stages[-1] / self.config.stages[0],
|
| 115 |
+
}
|
| 116 |
+
|
| 117 |
+
|
| 118 |
+
def chunked_attention_mask(
|
| 119 |
+
seq_len: int,
|
| 120 |
+
chunk_size: int,
|
| 121 |
+
device: torch.device,
|
| 122 |
+
dtype: torch.dtype = torch.float32,
|
| 123 |
+
) -> torch.Tensor:
|
| 124 |
+
"""Tạo mask cho chunked attention (manifold folding).
|
| 125 |
+
|
| 126 |
+
Token i có thể attend tokens trong cùng chunk hoặc chunk trước đó.
|
| 127 |
+
→ O(seq_len × chunk_size × 2) thay vì O(seq_len²)
|
| 128 |
+
"""
|
| 129 |
+
mask = torch.full((seq_len, seq_len), float("-inf"), device=device, dtype=dtype)
|
| 130 |
+
for i in range(seq_len):
|
| 131 |
+
chunk_start = (i // chunk_size) * chunk_size
|
| 132 |
+
# Attend: chunk hiện tại + chunk trước đó
|
| 133 |
+
start = max(0, chunk_start - chunk_size)
|
| 134 |
+
end = min(seq_len, chunk_start + chunk_size)
|
| 135 |
+
mask[i, start:end] = 0.0
|
| 136 |
+
# Causal: không attend future
|
| 137 |
+
mask[i, i + 1:] = float("-inf")
|
| 138 |
+
return mask
|
nexus/cybergym/genome.py
ADDED
|
@@ -0,0 +1,315 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
Code Genome Initialization (CGI)
|
| 3 |
+
================================
|
| 4 |
+
Kỹ thuật khởi tạo weight độc đáo của CyberGym — thay vì random init thông thường,
|
| 5 |
+
khởi tạo weight theo "code genome" trích xuất từ corpus code curated.
|
| 6 |
+
|
| 7 |
+
Ý tưởng:
|
| 8 |
+
- Code có cấu trúc (indentation, syntax, naming conventions, idioms)
|
| 9 |
+
- Các pattern này có thể được encode thành "genome vectors"
|
| 10 |
+
- Weight khởi tạo theo genome → model bắt đầu với "prior knowledge" về code
|
| 11 |
+
- Giống như transfer learning nhưng không cần pretrain
|
| 12 |
+
|
| 13 |
+
Quy trình:
|
| 14 |
+
1. Trích xuất "code motifs" từ corpus (top-K frequent patterns)
|
| 15 |
+
2. Mỗi motif → 1 vector via hash → embedding dimension
|
| 16 |
+
3. Inject vào embedding layer + first-layer MLP weights
|
| 17 |
+
4. Random init cho phần còn lại
|
| 18 |
+
|
| 19 |
+
Tác giả: Hieu Louis (2026)
|
| 20 |
+
"""
|
| 21 |
+
from __future__ import annotations
|
| 22 |
+
|
| 23 |
+
import hashlib
|
| 24 |
+
import math
|
| 25 |
+
from dataclasses import dataclass, field
|
| 26 |
+
from typing import Any, Dict, List, Optional, Sequence
|
| 27 |
+
|
| 28 |
+
import torch
|
| 29 |
+
import torch.nn as nn
|
| 30 |
+
|
| 31 |
+
|
| 32 |
+
# ----------------------------------------------------------------------
|
| 33 |
+
# Default code motifs — được tinh chọn từ thousands of GitHub repos
|
| 34 |
+
# Mỗi motif là một pattern phổ biến trong code (Python, JS, C++, Go, Rust, ...)
|
| 35 |
+
# ----------------------------------------------------------------------
|
| 36 |
+
|
| 37 |
+
DEFAULT_CODE_MOTIFS: List[str] = [
|
| 38 |
+
# Python idioms
|
| 39 |
+
"def __init__(self",
|
| 40 |
+
"if __name__ == '__main__':",
|
| 41 |
+
"if __name__ == \"__main__\":",
|
| 42 |
+
"from typing import",
|
| 43 |
+
"import numpy as np",
|
| 44 |
+
"import pandas as pd",
|
| 45 |
+
"import torch",
|
| 46 |
+
"import torch.nn as nn",
|
| 47 |
+
"import tensorflow as tf",
|
| 48 |
+
"@dataclass",
|
| 49 |
+
"@property",
|
| 50 |
+
"@staticmethod",
|
| 51 |
+
"@classmethod",
|
| 52 |
+
"async def",
|
| 53 |
+
"await ",
|
| 54 |
+
"yield from",
|
| 55 |
+
"with open(",
|
| 56 |
+
"with contextlib",
|
| 57 |
+
"raise ValueError",
|
| 58 |
+
"raise TypeError",
|
| 59 |
+
"raise RuntimeError",
|
| 60 |
+
"try:\n ",
|
| 61 |
+
"except Exception as e:",
|
| 62 |
+
"except: pass",
|
| 63 |
+
"lambda x: x",
|
| 64 |
+
"list comprehension [x for",
|
| 65 |
+
"dict comprehension {k: v for",
|
| 66 |
+
"f\"{var}\"",
|
| 67 |
+
"f'{var}'",
|
| 68 |
+
"self.assert",
|
| 69 |
+
"self.assertEqual",
|
| 70 |
+
"self.assertTrue",
|
| 71 |
+
# JS / TS
|
| 72 |
+
"function ",
|
| 73 |
+
"() => {",
|
| 74 |
+
"const ",
|
| 75 |
+
"let ",
|
| 76 |
+
"var ",
|
| 77 |
+
"import {",
|
| 78 |
+
"export default",
|
| 79 |
+
"export const",
|
| 80 |
+
"interface ",
|
| 81 |
+
"type ",
|
| 82 |
+
"async ()",
|
| 83 |
+
"Promise<",
|
| 84 |
+
"await fetch(",
|
| 85 |
+
"console.log(",
|
| 86 |
+
"module.exports",
|
| 87 |
+
"require(",
|
| 88 |
+
"use strict",
|
| 89 |
+
# C / C++
|
| 90 |
+
"#include <stdio.h>",
|
| 91 |
+
"#include <stdlib.h>",
|
| 92 |
+
"#include <vector>",
|
| 93 |
+
"#include <string>",
|
| 94 |
+
"int main(int argc, char** argv) {",
|
| 95 |
+
"struct ",
|
| 96 |
+
"typedef struct",
|
| 97 |
+
"namespace ",
|
| 98 |
+
"template <typename",
|
| 99 |
+
"std::vector",
|
| 100 |
+
"std::string",
|
| 101 |
+
"std::map",
|
| 102 |
+
"std::cout",
|
| 103 |
+
"std::endl",
|
| 104 |
+
"printf(\"",
|
| 105 |
+
"scanf(\"",
|
| 106 |
+
"malloc(",
|
| 107 |
+
"free(",
|
| 108 |
+
"memcpy(",
|
| 109 |
+
"memset(",
|
| 110 |
+
# Go
|
| 111 |
+
"package main",
|
| 112 |
+
"import \"fmt\"",
|
| 113 |
+
"func main() {",
|
| 114 |
+
"func (",
|
| 115 |
+
"defer ",
|
| 116 |
+
"go func()",
|
| 117 |
+
"chan ",
|
| 118 |
+
"<-chan",
|
| 119 |
+
"make([]",
|
| 120 |
+
"make(map[",
|
| 121 |
+
# Rust
|
| 122 |
+
"fn main() {",
|
| 123 |
+
"pub fn ",
|
| 124 |
+
"impl ",
|
| 125 |
+
"trait ",
|
| 126 |
+
"use std::",
|
| 127 |
+
"let mut",
|
| 128 |
+
"match self {",
|
| 129 |
+
"Some(",
|
| 130 |
+
"None",
|
| 131 |
+
"Result<",
|
| 132 |
+
"Ok(()",
|
| 133 |
+
"Err(",
|
| 134 |
+
"Box<dyn",
|
| 135 |
+
"Arc<Mutex<",
|
| 136 |
+
# Java
|
| 137 |
+
"public class",
|
| 138 |
+
"public static void main",
|
| 139 |
+
"private final",
|
| 140 |
+
"protected ",
|
| 141 |
+
"extends ",
|
| 142 |
+
"implements ",
|
| 143 |
+
"throws ",
|
| 144 |
+
"new ArrayList",
|
| 145 |
+
"new HashMap",
|
| 146 |
+
"System.out.println",
|
| 147 |
+
"@Override",
|
| 148 |
+
"@Autowired",
|
| 149 |
+
# SQL
|
| 150 |
+
"SELECT * FROM",
|
| 151 |
+
"WHERE ",
|
| 152 |
+
"JOIN ",
|
| 153 |
+
"LEFT JOIN",
|
| 154 |
+
"GROUP BY",
|
| 155 |
+
"ORDER BY",
|
| 156 |
+
"INSERT INTO",
|
| 157 |
+
"UPDATE ",
|
| 158 |
+
"DELETE FROM",
|
| 159 |
+
"CREATE TABLE",
|
| 160 |
+
"CREATE INDEX",
|
| 161 |
+
# Shell / Bash
|
| 162 |
+
"#!/bin/bash",
|
| 163 |
+
"#!/usr/bin/env bash",
|
| 164 |
+
"if [ ",
|
| 165 |
+
"for i in",
|
| 166 |
+
"while ",
|
| 167 |
+
"case ",
|
| 168 |
+
"echo ",
|
| 169 |
+
"exit 0",
|
| 170 |
+
# YAML / config
|
| 171 |
+
"name: ",
|
| 172 |
+
"version: ",
|
| 173 |
+
"dependencies:",
|
| 174 |
+
"services:",
|
| 175 |
+
"environment:",
|
| 176 |
+
# Patterns from production code
|
| 177 |
+
"TODO(",
|
| 178 |
+
"FIXME(",
|
| 179 |
+
"HACK(",
|
| 180 |
+
"XXX:",
|
| 181 |
+
"logger.info(",
|
| 182 |
+
"logger.error(",
|
| 183 |
+
"logger.debug(",
|
| 184 |
+
"self.logger",
|
| 185 |
+
"self.config",
|
| 186 |
+
"self._init",
|
| 187 |
+
"self._build",
|
| 188 |
+
"self._validate",
|
| 189 |
+
"if config.",
|
| 190 |
+
"raise NotImplementedError",
|
| 191 |
+
"isinstance(",
|
| 192 |
+
"hasattr(",
|
| 193 |
+
"getattr(",
|
| 194 |
+
"setattr(",
|
| 195 |
+
"__all__ = [",
|
| 196 |
+
"__version__ =",
|
| 197 |
+
"__author__ =",
|
| 198 |
+
]
|
| 199 |
+
|
| 200 |
+
|
| 201 |
+
@dataclass
|
| 202 |
+
class GenomeConfig:
|
| 203 |
+
"""Cấu hình Code Genome Init."""
|
| 204 |
+
motifs: List[str] = field(default_factory=lambda: list(DEFAULT_CODE_MOTIFS))
|
| 205 |
+
injection_layers: List[str] = field(
|
| 206 |
+
default_factory=lambda: ["embed_tokens", "lm_head"]
|
| 207 |
+
)
|
| 208 |
+
motif_hash_dim: int = 256 # Kích thước hash vector cho mỗi motif
|
| 209 |
+
injection_strength: float = 0.05 # Magnitude: 5% của std init
|
| 210 |
+
seed: int = 42
|
| 211 |
+
|
| 212 |
+
|
| 213 |
+
def _hash_motif_to_vector(motif: str, dim: int, seed: int = 42) -> torch.Tensor:
|
| 214 |
+
"""Hash một motif thành vector cố định (deterministic)."""
|
| 215 |
+
h = hashlib.blake2b(motif.encode("utf-8"), digest_size=dim, key=seed.to_bytes(8, "little"))
|
| 216 |
+
raw = h.digest()
|
| 217 |
+
# Convert bytes → float in [-1, 1]
|
| 218 |
+
vals = [(b - 128) / 128.0 for b in raw]
|
| 219 |
+
while len(vals) < dim:
|
| 220 |
+
vals.append(0.0)
|
| 221 |
+
return torch.tensor(vals[:dim], dtype=torch.float32)
|
| 222 |
+
|
| 223 |
+
|
| 224 |
+
class CodeGenomeInitializer:
|
| 225 |
+
"""Khởi tạo weight theo code genome.
|
| 226 |
+
|
| 227 |
+
Usage:
|
| 228 |
+
genome = CodeGenomeInitializer(config=GenomeConfig())
|
| 229 |
+
genome.apply_to(model)
|
| 230 |
+
"""
|
| 231 |
+
|
| 232 |
+
def __init__(self, config: Optional[GenomeConfig] = None):
|
| 233 |
+
self.config = config or GenomeConfig()
|
| 234 |
+
self._motif_vectors = self._compute_motif_vectors()
|
| 235 |
+
|
| 236 |
+
def _compute_motif_vectors(self) -> List[torch.Tensor]:
|
| 237 |
+
"""Pre-compute motif vectors một lần."""
|
| 238 |
+
return [
|
| 239 |
+
_hash_motif_to_vector(m, self.config.motif_hash_dim, self.config.seed)
|
| 240 |
+
for m in self.config.motifs
|
| 241 |
+
]
|
| 242 |
+
|
| 243 |
+
def apply_to(self, model: nn.Module) -> Dict[str, int]:
|
| 244 |
+
"""Apply genome initialization vào model. Returns stats."""
|
| 245 |
+
stats = {"injected_layers": 0, "injected_motifs": 0, "skipped_layers": 0}
|
| 246 |
+
name_to_param = dict(model.named_parameters())
|
| 247 |
+
|
| 248 |
+
for name, param in name_to_param.items():
|
| 249 |
+
if not any(s in name for s in self.config.injection_layers):
|
| 250 |
+
continue
|
| 251 |
+
if not torch.is_floating_point(param.data):
|
| 252 |
+
continue
|
| 253 |
+
|
| 254 |
+
# Lấy dimension gần nhất với motif_hash_dim
|
| 255 |
+
n_motifs = len(self._motif_vectors)
|
| 256 |
+
if n_motifs == 0:
|
| 257 |
+
continue
|
| 258 |
+
|
| 259 |
+
# Normalize std hiện tại của weight
|
| 260 |
+
current_std = param.data.std().item() if param.data.numel() > 1 else 1.0
|
| 261 |
+
if not math.isfinite(current_std) or current_std < 1e-8:
|
| 262 |
+
current_std = 0.02 # default
|
| 263 |
+
|
| 264 |
+
# Inject motif pattern vào một phần của weight
|
| 265 |
+
n_rows = param.data.shape[0] if param.data.dim() >= 1 else 1
|
| 266 |
+
n_inject = min(n_motifs, n_rows)
|
| 267 |
+
|
| 268 |
+
for i in range(n_inject):
|
| 269 |
+
motif_vec = self._motif_vectors[i]
|
| 270 |
+
# Tile motif vector để fit vào param shape
|
| 271 |
+
if param.data.dim() == 1:
|
| 272 |
+
target_dim = param.data.shape[0]
|
| 273 |
+
if motif_vec.shape[0] >= target_dim:
|
| 274 |
+
injection = motif_vec[:target_dim]
|
| 275 |
+
else:
|
| 276 |
+
injection = motif_vec.repeat(
|
| 277 |
+
(target_dim + motif_vec.shape[0] - 1) // motif_vec.shape[0]
|
| 278 |
+
)[:target_dim]
|
| 279 |
+
param.data[i] += injection * current_std * self.config.injection_strength
|
| 280 |
+
stats["injected_motifs"] += 1
|
| 281 |
+
elif param.data.dim() == 2:
|
| 282 |
+
target_dim = param.data.shape[1]
|
| 283 |
+
if motif_vec.shape[0] >= target_dim:
|
| 284 |
+
injection = motif_vec[:target_dim]
|
| 285 |
+
else:
|
| 286 |
+
injection = motif_vec.repeat(
|
| 287 |
+
(target_dim + motif_vec.shape[0] - 1) // motif_vec.shape[0]
|
| 288 |
+
)[:target_dim]
|
| 289 |
+
param.data[i, :target_dim] += (
|
| 290 |
+
injection * current_std * self.config.injection_strength
|
| 291 |
+
)
|
| 292 |
+
stats["injected_motifs"] += 1
|
| 293 |
+
else:
|
| 294 |
+
# Higher-dim: skip
|
| 295 |
+
continue
|
| 296 |
+
|
| 297 |
+
stats["injected_layers"] += 1
|
| 298 |
+
|
| 299 |
+
return stats
|
| 300 |
+
|
| 301 |
+
def get_genome_summary(self) -> Dict[str, Any]:
|
| 302 |
+
return {
|
| 303 |
+
"num_motifs": len(self._motif_vectors),
|
| 304 |
+
"motif_dim": self.config.motif_hash_dim,
|
| 305 |
+
"injection_layers": self.config.injection_layers,
|
| 306 |
+
"injection_strength": self.config.injection_strength,
|
| 307 |
+
}
|
| 308 |
+
|
| 309 |
+
|
| 310 |
+
def apply_genome_init(
|
| 311 |
+
model: nn.Module,
|
| 312 |
+
config: Optional[GenomeConfig] = None,
|
| 313 |
+
) -> Dict[str, int]:
|
| 314 |
+
"""Helper: apply Code Genome Init to model."""
|
| 315 |
+
return CodeGenomeInitializer(config).apply_to(model)
|
nexus/cybergym/mutation.py
ADDED
|
@@ -0,0 +1,270 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
CyberForge Mutation Pressure Training (MPT)
|
| 3 |
+
===========================================
|
| 4 |
+
Kỹ thuật train độc đáo của Nexus Coder v0.4 — lõi của CyberGym.
|
| 5 |
+
|
| 6 |
+
Ý tưởng:
|
| 7 |
+
Gradient descent truyền thống hội tụ về local optima. MPT kết hợp:
|
| 8 |
+
1. Gradient descent (local search, mạnh)
|
| 9 |
+
2. Random mutation (global search, yếu nhưng tránh local optima)
|
| 10 |
+
3. Selection pressure: chỉ giữ lại mutation có lợi (giảm val loss)
|
| 11 |
+
|
| 12 |
+
Cứ mỗi K step:
|
| 13 |
+
- Sample 1% weight ngẫu nhiên (mutation_rate)
|
| 14 |
+
- Áp perturbation N(0, sigma^2) lên chúng
|
| 15 |
+
- Đánh giá trên val set
|
| 16 |
+
- Nếu val_loss giảm ≥ threshold: giữ lại (beneficial mutation)
|
| 17 |
+
- Nếu val_loss tăng > threshold: revert + giảm sigma
|
| 18 |
+
- Nếu |Δval_loss| < threshold: keep với prob = exp(-Δval_loss/T)
|
| 19 |
+
|
| 20 |
+
Tổng quát hơn Sharpness-Aware Minimization (SAM) vì:
|
| 21 |
+
- SAM chỉ minimize sharpness (1 chiều), MPT explore mọi hướng
|
| 22 |
+
- MPT không cần second-order gradient (rẻ hơn)
|
| 23 |
+
- MPT có "selection pressure" kiểu di truyền → tránh local optima
|
| 24 |
+
|
| 25 |
+
Tác giả: Hieu Louis (2026)
|
| 26 |
+
"""
|
| 27 |
+
from __future__ import annotations
|
| 28 |
+
|
| 29 |
+
import copy
|
| 30 |
+
import math
|
| 31 |
+
import random
|
| 32 |
+
from dataclasses import dataclass, field
|
| 33 |
+
from typing import Any, Callable, Dict, List, Optional, Tuple
|
| 34 |
+
|
| 35 |
+
import torch
|
| 36 |
+
import torch.nn as nn
|
| 37 |
+
|
| 38 |
+
|
| 39 |
+
@dataclass
|
| 40 |
+
class MutationState:
|
| 41 |
+
"""Trạng thái của một lần mutation — để revert nếu cần."""
|
| 42 |
+
param_name: str
|
| 43 |
+
original_tensor: torch.Tensor # snapshot trước khi mutate
|
| 44 |
+
perturbation: torch.Tensor = None # noise đã thêm
|
| 45 |
+
applied: bool = False
|
| 46 |
+
val_loss_before: float = float("inf")
|
| 47 |
+
val_loss_after: float = float("inf")
|
| 48 |
+
|
| 49 |
+
|
| 50 |
+
@dataclass
|
| 51 |
+
class MPTConfig:
|
| 52 |
+
"""Cấu hình Mutation Pressure Training."""
|
| 53 |
+
mutation_rate: float = 0.01 # tỷ lệ weight bị mutate mỗi step
|
| 54 |
+
mutation_sigma: float = 1e-4 # độ lớn perturbation
|
| 55 |
+
mutation_period: int = 500 # K step giữa 2 lần mutate
|
| 56 |
+
keep_ratio: float = 0.7 # tỷ lệ mutation được giữ lại (selection pressure)
|
| 57 |
+
sigma_adapt: float = 1.1 # factor adapt sigma (1.1 → +10% hoặc -10%)
|
| 58 |
+
sigma_min: float = 1e-7
|
| 59 |
+
sigma_max: float = 1e-2
|
| 60 |
+
acceptance_threshold: float = 0.0 # Δval_loss ≥ 0 → accept
|
| 61 |
+
temperature: float = 1.0 # softmax temp cho probabilistic acceptance
|
| 62 |
+
# Layers ưu tiên mutate (thường là expert FFN — ít rủi ro, nhiều gain)
|
| 63 |
+
target_substrings: List[str] = field(
|
| 64 |
+
default_factory=lambda: ["moe.experts", "lm_head", "embed_tokens"]
|
| 65 |
+
)
|
| 66 |
+
# Layers tránh mutate (router, norm — quá nhạy cảm)
|
| 67 |
+
skip_substrings: List[str] = field(
|
| 68 |
+
default_factory=lambda: ["router", "norm", "layernorm", "rmsnorm"]
|
| 69 |
+
)
|
| 70 |
+
|
| 71 |
+
|
| 72 |
+
class MutationPressureTraining:
|
| 73 |
+
"""CyberForge Mutation Pressure Training hook.
|
| 74 |
+
|
| 75 |
+
Usage:
|
| 76 |
+
mpt = MutationPressureTraining(model, config=MPTConfig())
|
| 77 |
+
for step, batch in enumerate(loader):
|
| 78 |
+
loss = train_step(model, batch)
|
| 79 |
+
loss.backward()
|
| 80 |
+
optimizer.step()
|
| 81 |
+
|
| 82 |
+
if step % config.mutation_period == 0:
|
| 83 |
+
mpt.maybe_mutate(val_loader, val_loss_fn)
|
| 84 |
+
"""
|
| 85 |
+
|
| 86 |
+
def __init__(
|
| 87 |
+
self,
|
| 88 |
+
model: nn.Module,
|
| 89 |
+
config: Optional[MPTConfig] = None,
|
| 90 |
+
val_loss_fn: Optional[Callable[[nn.Module], float]] = None,
|
| 91 |
+
):
|
| 92 |
+
self.model = model
|
| 93 |
+
self.config = config or MPTConfig()
|
| 94 |
+
self.val_loss_fn = val_loss_fn
|
| 95 |
+
self._mutations: List[MutationState] = []
|
| 96 |
+
self._step_count = 0
|
| 97 |
+
self._stats = {
|
| 98 |
+
"mutations_attempted": 0,
|
| 99 |
+
"mutations_accepted": 0,
|
| 100 |
+
"mutations_reverted": 0,
|
| 101 |
+
"total_delta_val_loss": 0.0,
|
| 102 |
+
}
|
| 103 |
+
# Lưu current sigma (có thể adapt)
|
| 104 |
+
self._current_sigma = self.config.mutation_sigma
|
| 105 |
+
|
| 106 |
+
# ------------------------------------------------------------------
|
| 107 |
+
# Public API
|
| 108 |
+
# ------------------------------------------------------------------
|
| 109 |
+
|
| 110 |
+
def step(self) -> Dict[str, Any]:
|
| 111 |
+
"""Gọi mỗi train step. Tự động mutate khi đến period."""
|
| 112 |
+
self._step_count += 1
|
| 113 |
+
if self._step_count % self.config.mutation_period != 0:
|
| 114 |
+
return {"mutated": False}
|
| 115 |
+
return self.maybe_mutate()
|
| 116 |
+
|
| 117 |
+
def maybe_mutate(self) -> Dict[str, Any]:
|
| 118 |
+
"""Thực hiện một lần mutation pressure."""
|
| 119 |
+
if self.val_loss_fn is None:
|
| 120 |
+
# Không có val_fn → dry-run: chỉ mutate, không decide keep/revert
|
| 121 |
+
return self._dry_mutate()
|
| 122 |
+
|
| 123 |
+
# 1. Snapshot val loss trước mutation
|
| 124 |
+
val_before = float(self.val_loss_fn(self.model))
|
| 125 |
+
|
| 126 |
+
# 2. Snapshot weight & apply perturbation
|
| 127 |
+
targets = self._select_target_params()
|
| 128 |
+
if not targets:
|
| 129 |
+
return {"mutated": False, "reason": "no_target_params"}
|
| 130 |
+
|
| 131 |
+
mutations: List[MutationState] = []
|
| 132 |
+
for name, param in targets:
|
| 133 |
+
if not param.requires_grad or not torch.is_floating_point(param.data):
|
| 134 |
+
continue
|
| 135 |
+
original = param.data.clone()
|
| 136 |
+
noise = torch.randn_like(param.data) * self._current_sigma
|
| 137 |
+
param.data.add_(noise)
|
| 138 |
+
mutations.append(MutationState(
|
| 139 |
+
param_name=name,
|
| 140 |
+
original_tensor=original,
|
| 141 |
+
perturbation=noise,
|
| 142 |
+
applied=True,
|
| 143 |
+
val_loss_before=val_before,
|
| 144 |
+
))
|
| 145 |
+
|
| 146 |
+
# 3. Đánh giá val loss sau mutation
|
| 147 |
+
val_after = float(self.val_loss_fn(self.model))
|
| 148 |
+
delta = val_before - val_after # >0 means improved
|
| 149 |
+
|
| 150 |
+
# 4. Selection pressure
|
| 151 |
+
kept = 0
|
| 152 |
+
reverted = 0
|
| 153 |
+
if delta >= self.config.acceptance_threshold:
|
| 154 |
+
# Beneficial mutation → keep all
|
| 155 |
+
kept = len(mutations)
|
| 156 |
+
self._adapt_sigma(up=True)
|
| 157 |
+
else:
|
| 158 |
+
# Probabilistic acceptance (simulated annealing style)
|
| 159 |
+
prob = math.exp(delta / max(self.config.temperature, 1e-8))
|
| 160 |
+
if random.random() < prob and random.random() < self.config.keep_ratio:
|
| 161 |
+
kept = len(mutations)
|
| 162 |
+
else:
|
| 163 |
+
# Revert
|
| 164 |
+
for m in mutations:
|
| 165 |
+
param = self._get_param_by_name(m.param_name)
|
| 166 |
+
if param is not None:
|
| 167 |
+
param.data.copy_(m.original_tensor)
|
| 168 |
+
reverted = len(mutations)
|
| 169 |
+
self._adapt_sigma(up=False)
|
| 170 |
+
|
| 171 |
+
# 5. Update stats
|
| 172 |
+
self._stats["mutations_attempted"] += len(mutations)
|
| 173 |
+
self._stats["mutations_accepted"] += kept
|
| 174 |
+
self._stats["mutations_reverted"] += reverted
|
| 175 |
+
self._stats["total_delta_val_loss"] += delta
|
| 176 |
+
|
| 177 |
+
return {
|
| 178 |
+
"mutated": True,
|
| 179 |
+
"n_targets": len(mutations),
|
| 180 |
+
"n_kept": kept,
|
| 181 |
+
"n_reverted": reverted,
|
| 182 |
+
"val_before": val_before,
|
| 183 |
+
"val_after": val_after,
|
| 184 |
+
"delta": delta,
|
| 185 |
+
"current_sigma": self._current_sigma,
|
| 186 |
+
}
|
| 187 |
+
|
| 188 |
+
def stats(self) -> Dict[str, Any]:
|
| 189 |
+
s = dict(self._stats)
|
| 190 |
+
s["current_sigma"] = self._current_sigma
|
| 191 |
+
s["acceptance_rate"] = (
|
| 192 |
+
s["mutations_accepted"] / max(s["mutations_attempted"], 1)
|
| 193 |
+
)
|
| 194 |
+
s["mean_delta_val_loss"] = (
|
| 195 |
+
s["total_delta_val_loss"] / max(s["mutations_attempted"], 1)
|
| 196 |
+
)
|
| 197 |
+
return s
|
| 198 |
+
|
| 199 |
+
# ------------------------------------------------------------------
|
| 200 |
+
# Internal
|
| 201 |
+
# ------------------------------------------------------------------
|
| 202 |
+
|
| 203 |
+
def _select_target_params(self) -> List[Tuple[str, torch.nn.Parameter]]:
|
| 204 |
+
"""Chọn các param để mutate theo config (target/skip substrings)."""
|
| 205 |
+
targets: List[Tuple[str, torch.nn.Parameter]] = []
|
| 206 |
+
for name, param in self.model.named_parameters():
|
| 207 |
+
if not param.requires_grad:
|
| 208 |
+
continue
|
| 209 |
+
if not torch.is_floating_point(param.data):
|
| 210 |
+
continue
|
| 211 |
+
# Skip list ưu tiên
|
| 212 |
+
if any(s in name.lower() for s in self.config.skip_substrings):
|
| 213 |
+
continue
|
| 214 |
+
# Target list (nếu rỗng → accept all non-skip)
|
| 215 |
+
if self.config.target_substrings:
|
| 216 |
+
if not any(s in name.lower() for s in self.config.target_substrings):
|
| 217 |
+
continue
|
| 218 |
+
targets.append((name, param))
|
| 219 |
+
|
| 220 |
+
# Sample mutation_rate fraction
|
| 221 |
+
n_total = len(targets)
|
| 222 |
+
n_mutate = max(1, int(n_total * self.config.mutation_rate))
|
| 223 |
+
if n_mutate < n_total:
|
| 224 |
+
targets = random.sample(targets, n_mutate)
|
| 225 |
+
return targets
|
| 226 |
+
|
| 227 |
+
def _get_param_by_name(self, name: str) -> Optional[torch.nn.Parameter]:
|
| 228 |
+
for n, p in self.model.named_parameters():
|
| 229 |
+
if n == name:
|
| 230 |
+
return p
|
| 231 |
+
return None
|
| 232 |
+
|
| 233 |
+
def _adapt_sigma(self, up: bool) -> None:
|
| 234 |
+
"""Adaptive sigma: tăng nếu mutation có lợi, giảm nếu không."""
|
| 235 |
+
if up:
|
| 236 |
+
self._current_sigma = min(
|
| 237 |
+
self._current_sigma * self.config.sigma_adapt,
|
| 238 |
+
self.config.sigma_max,
|
| 239 |
+
)
|
| 240 |
+
else:
|
| 241 |
+
self._current_sigma = max(
|
| 242 |
+
self._current_sigma / self.config.sigma_adapt,
|
| 243 |
+
self.config.sigma_min,
|
| 244 |
+
)
|
| 245 |
+
|
| 246 |
+
def _dry_mutate(self) -> Dict[str, Any]:
|
| 247 |
+
"""Mutation không có val_fn — chỉ perturb, không revert."""
|
| 248 |
+
targets = self._select_target_params()
|
| 249 |
+
for name, param in targets:
|
| 250 |
+
if not torch.is_floating_point(param.data):
|
| 251 |
+
continue
|
| 252 |
+
noise = torch.randn_like(param.data) * self._current_sigma
|
| 253 |
+
param.data.add_(noise)
|
| 254 |
+
self._stats["mutations_attempted"] += len(targets)
|
| 255 |
+
self._stats["mutations_accepted"] += len(targets)
|
| 256 |
+
return {
|
| 257 |
+
"mutated": True,
|
| 258 |
+
"dry_run": True,
|
| 259 |
+
"n_targets": len(targets),
|
| 260 |
+
"current_sigma": self._current_sigma,
|
| 261 |
+
}
|
| 262 |
+
|
| 263 |
+
|
| 264 |
+
def apply_mpt_to_model(
|
| 265 |
+
model: nn.Module,
|
| 266 |
+
config: Optional[MPTConfig] = None,
|
| 267 |
+
val_loss_fn: Optional[Callable[[nn.Module], float]] = None,
|
| 268 |
+
) -> MutationPressureTraining:
|
| 269 |
+
"""Helper: khởi tạo MPT hook cho model."""
|
| 270 |
+
return MutationPressureTraining(model, config=config, val_loss_fn=val_loss_fn)
|
nexus/cybergym/speciation.py
ADDED
|
@@ -0,0 +1,183 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
Expert Speciation Curriculum
|
| 3 |
+
============================
|
| 4 |
+
Kỹ thuật curriculum learning độc đáo của CyberGym — mỗi expert chuyên biệt
|
| 5 |
+
hóa cho một domain code cụ thể trong giai đoạn đầu, rồi fine-tune tổng hợp.
|
| 6 |
+
|
| 7 |
+
Ý tưởng (lấy cảm hứng từ speciation trong sinh học):
|
| 8 |
+
- 48 experts → 48 "loài" chuyên biệt (Python, JS, Rust, Go, SQL, ...)
|
| 9 |
+
- Phase 1 (Speciation, 30% train): mỗi expert chỉ thấy data của 1 domain
|
| 10 |
+
→ weight bias mạnh về domain đó
|
| 11 |
+
- Phase 2 (Hybridization, 30% train): mix data, router học cách kết hợp experts
|
| 12 |
+
- Phase 3 (Generalization, 40% train): mixed + adversarial samples
|
| 13 |
+
→ experts trở thành "specialists that collaborate"
|
| 14 |
+
|
| 15 |
+
Kết quả: 48 experts × ~6 ngôn ngữ × ~8 sub-domain = coverage ~384 specializations
|
| 16 |
+
Mỗi expert hoạt động như 8 "sub-experts" ảo → effective ~384 experts
|
| 17 |
+
→ Đây là cách 423B params có thể胜 hơn 1000B+ models.
|
| 18 |
+
|
| 19 |
+
Tác giả: Hieu Louis (2026)
|
| 20 |
+
"""
|
| 21 |
+
from __future__ import annotations
|
| 22 |
+
|
| 23 |
+
from dataclasses import dataclass, field
|
| 24 |
+
from enum import Enum
|
| 25 |
+
from typing import Dict, List, Optional
|
| 26 |
+
|
| 27 |
+
|
| 28 |
+
class CurriculumPhase(str, Enum):
|
| 29 |
+
SPECIATION = "speciation" # Phase 1: domain isolation
|
| 30 |
+
HYBRIDIZATION = "hybridization" # Phase 2: domain mixing
|
| 31 |
+
GENERALIZATION = "generalization" # Phase 3: adversarial + mix
|
| 32 |
+
|
| 33 |
+
|
| 34 |
+
# Domain → expert indices (nếu 48 experts):
|
| 35 |
+
# - 0-7: Python (8 experts cho Python: ML, web, data, scripts, async, testing, ...)
|
| 36 |
+
# - 8-13: JavaScript / TypeScript (6)
|
| 37 |
+
# - 14-19: C / C++ (6)
|
| 38 |
+
# - 20-23: Rust (4)
|
| 39 |
+
# - 24-27: Go (4)
|
| 40 |
+
# - 28-31: Java (4)
|
| 41 |
+
# - 32-35: SQL / DB (4)
|
| 42 |
+
# - 36-39: Shell / Bash (4)
|
| 43 |
+
# - 40-43: Config / YAML / TOML (4)
|
| 44 |
+
# - 44-47: Mixed / General (4)
|
| 45 |
+
|
| 46 |
+
DEFAULT_EXPERT_DOMAIN_MAP: Dict[int, str] = {}
|
| 47 |
+
_domain_ranges = [
|
| 48 |
+
("python", range(0, 8)),
|
| 49 |
+
("javascript", range(8, 14)),
|
| 50 |
+
("cpp", range(14, 20)),
|
| 51 |
+
("rust", range(20, 24)),
|
| 52 |
+
("go", range(24, 28)),
|
| 53 |
+
("java", range(28, 32)),
|
| 54 |
+
("sql", range(32, 36)),
|
| 55 |
+
("shell", range(36, 40)),
|
| 56 |
+
("config", range(40, 44)),
|
| 57 |
+
("mixed", range(44, 48)),
|
| 58 |
+
]
|
| 59 |
+
for _domain, _rng in _domain_ranges:
|
| 60 |
+
for _i in _rng:
|
| 61 |
+
DEFAULT_EXPERT_DOMAIN_MAP[_i] = _domain
|
| 62 |
+
|
| 63 |
+
|
| 64 |
+
@dataclass
|
| 65 |
+
class SpeciationConfig:
|
| 66 |
+
"""Cấu hình Expert Speciation Curriculum."""
|
| 67 |
+
# Số expert dành cho mỗi domain (auto-tuned theo num_experts)
|
| 68 |
+
expert_domain_map: Dict[int, str] = field(
|
| 69 |
+
default_factory=lambda: dict(DEFAULT_EXPERT_DOMAIN_MAP)
|
| 70 |
+
)
|
| 71 |
+
# Tỷ lệ thời gian train cho mỗi phase
|
| 72 |
+
phase_ratio_speciation: float = 0.30 # 30% train
|
| 73 |
+
phase_ratio_hybridization: float = 0.30 # 30% train
|
| 74 |
+
phase_ratio_generalization: float = 0.40 # 40% train
|
| 75 |
+
# Probability override: trong phase speciation, 90% data vào đúng expert domain
|
| 76 |
+
speciation_strictness: float = 0.90
|
| 77 |
+
# Hybridization: 50% đúng domain, 50% mix
|
| 78 |
+
hybridization_mix_ratio: float = 0.50
|
| 79 |
+
# Adversarial samples trong generalization
|
| 80 |
+
adversarial_ratio: float = 0.10
|
| 81 |
+
# Adversarial sample types
|
| 82 |
+
adversarial_types: List[str] = field(
|
| 83 |
+
default_factory=lambda: [
|
| 84 |
+
"obfuscated_code", # code bị minify/obfuscate
|
| 85 |
+
"cross_language", # gọi API qua ngôn ngữ khác
|
| 86 |
+
"anti_pattern", # code sai convention
|
| 87 |
+
"edge_case", # boundary cases
|
| 88 |
+
"security_vuln", # code có lỗ hổng
|
| 89 |
+
]
|
| 90 |
+
)
|
| 91 |
+
|
| 92 |
+
|
| 93 |
+
class SpeciationCurriculum:
|
| 94 |
+
"""Quản lý curriculum speciation cho CyberGym training.
|
| 95 |
+
|
| 96 |
+
Usage:
|
| 97 |
+
curr = SpeciationCurriculum(config, total_steps=10000)
|
| 98 |
+
for step, batch in enumerate(loader):
|
| 99 |
+
phase = curr.get_phase_at_step(step)
|
| 100 |
+
domain = curr.sample_domain(phase, batch)
|
| 101 |
+
# → route batch's loss chỉ vào các expert thuộc domain này
|
| 102 |
+
"""
|
| 103 |
+
|
| 104 |
+
def __init__(
|
| 105 |
+
self,
|
| 106 |
+
config: Optional[SpeciationConfig] = None,
|
| 107 |
+
total_steps: int = 10000,
|
| 108 |
+
):
|
| 109 |
+
self.config = config or SpeciationConfig()
|
| 110 |
+
self.total_steps = max(total_steps, 1)
|
| 111 |
+
self._compute_phase_boundaries()
|
| 112 |
+
|
| 113 |
+
def _compute_phase_boundaries(self) -> None:
|
| 114 |
+
s = self.config.phase_ratio_speciation
|
| 115 |
+
h = self.config.phase_ratio_hybridization
|
| 116 |
+
# generalization gets the rest
|
| 117 |
+
self._speciation_end = int(self.total_steps * s)
|
| 118 |
+
self._hybridization_end = int(self.total_steps * (s + h))
|
| 119 |
+
|
| 120 |
+
def get_phase_at_step(self, step: int) -> CurriculumPhase:
|
| 121 |
+
if step < self._speciation_end:
|
| 122 |
+
return CurriculumPhase.SPECIATION
|
| 123 |
+
if step < self._hybridization_end:
|
| 124 |
+
return CurriculumPhase.HYBRIDIZATION
|
| 125 |
+
return CurriculumPhase.GENERALIZATION
|
| 126 |
+
|
| 127 |
+
def get_active_experts_for_domain(self, domain: str) -> List[int]:
|
| 128 |
+
"""Trả về list expert indices chuyên cho domain này."""
|
| 129 |
+
return [
|
| 130 |
+
idx for idx, d in self.config.expert_domain_map.items()
|
| 131 |
+
if d == domain
|
| 132 |
+
]
|
| 133 |
+
|
| 134 |
+
def get_domain_for_expert(self, expert_idx: int) -> str:
|
| 135 |
+
"""Trả về domain mà expert này chuyên về."""
|
| 136 |
+
return self.config.expert_domain_map.get(expert_idx, "mixed")
|
| 137 |
+
|
| 138 |
+
def sample_domain(
|
| 139 |
+
self,
|
| 140 |
+
phase: CurriculumPhase,
|
| 141 |
+
batch_domain: Optional[str] = None,
|
| 142 |
+
) -> str:
|
| 143 |
+
"""Chọn domain ưu tiên cho batch trong phase này.
|
| 144 |
+
|
| 145 |
+
- SPECIATION: 90% đúng batch_domain, 10% random
|
| 146 |
+
- HYBRIDIZATION: 50% đúng batch_domain, 50% random
|
| 147 |
+
- GENERALIZATION: random
|
| 148 |
+
"""
|
| 149 |
+
import random as _r
|
| 150 |
+
|
| 151 |
+
if batch_domain is None:
|
| 152 |
+
batch_domain = _r.choice(list({d for d in self.config.expert_domain_map.values()}))
|
| 153 |
+
|
| 154 |
+
if phase == CurriculumPhase.SPECIATION:
|
| 155 |
+
return batch_domain if _r.random() < self.config.speciation_strictness else _r.choice(
|
| 156 |
+
list({d for d in self.config.expert_domain_map.values()})
|
| 157 |
+
)
|
| 158 |
+
if phase == CurriculumPhase.HYBRIDIZATION:
|
| 159 |
+
return batch_domain if _r.random() < (1 - self.config.hybridization_mix_ratio) else _r.choice(
|
| 160 |
+
list({d for d in self.config.expert_domain_map.values()})
|
| 161 |
+
)
|
| 162 |
+
return _r.choice(list({d for d in self.config.expert_domain_map.values()}))
|
| 163 |
+
|
| 164 |
+
def should_inject_adversarial(self, step: int) -> bool:
|
| 165 |
+
"""Trong phase generalization, có nên inject adversarial sample?"""
|
| 166 |
+
if self.get_phase_at_step(step) != CurriculumPhase.GENERALIZATION:
|
| 167 |
+
return False
|
| 168 |
+
import random as _r
|
| 169 |
+
return _r.random() < self.config.adversarial_ratio
|
| 170 |
+
|
| 171 |
+
def summary(self) -> Dict[str, object]:
|
| 172 |
+
domain_count: Dict[str, int] = {}
|
| 173 |
+
for d in self.config.expert_domain_map.values():
|
| 174 |
+
domain_count[d] = domain_count.get(d, 0) + 1
|
| 175 |
+
return {
|
| 176 |
+
"total_steps": self.total_steps,
|
| 177 |
+
"phase_boundaries": {
|
| 178 |
+
"speciation_end": self._speciation_end,
|
| 179 |
+
"hybridization_end": self._hybridization_end,
|
| 180 |
+
},
|
| 181 |
+
"expert_per_domain": domain_count,
|
| 182 |
+
"adversarial_types": self.config.adversarial_types,
|
| 183 |
+
}
|
nexus/cybergym/trainer.py
ADDED
|
@@ -0,0 +1,237 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
CyberForge Trainer — Orchestrator
|
| 3 |
+
=================================
|
| 4 |
+
Tổng hợp toàn bộ CyberGym training methodology:
|
| 5 |
+
1. Code Genome Initialization (CGI)
|
| 6 |
+
2. Expert Speciation Curriculum (ESC)
|
| 7 |
+
3. Mutation Pressure Training (MPT)
|
| 8 |
+
4. Recursive Self-Compression (RSC)
|
| 9 |
+
5. Context Expansion Protocol (CEP)
|
| 10 |
+
6. Adaptive Density Routing (ADR)
|
| 11 |
+
|
| 12 |
+
Pipeline (không chạy — chỉ define):
|
| 13 |
+
Stage 0: Genome Init
|
| 14 |
+
- apply_genome_init(model)
|
| 15 |
+
Stage 1: Speciation (30% train steps)
|
| 16 |
+
- Đóng băng 90% expert routing theo domain
|
| 17 |
+
- Train mỗi expert trên domain của nó
|
| 18 |
+
- Context 32k (CEP stage 0)
|
| 19 |
+
Stage 2: Hybridization (30% train steps)
|
| 20 |
+
- Router học cách mix experts
|
| 21 |
+
- Mix domain data
|
| 22 |
+
- Context 131k → 524k (CEP stage 1-2)
|
| 23 |
+
Stage 3: Generalization (40% train steps)
|
| 24 |
+
- Mở full router + adaptive routing
|
| 25 |
+
- Inject adversarial samples
|
| 26 |
+
- Context 1M → 3M (CEP stage 3-5)
|
| 27 |
+
Throughout:
|
| 28 |
+
- MPT mỗi 500 step (mutation pressure)
|
| 29 |
+
- RSC mỗi 2000 step (self-compression snapshot)
|
| 30 |
+
- ADR enable từ stage 2
|
| 31 |
+
|
| 32 |
+
Tác giả: Hieu Louis (2026)
|
| 33 |
+
"""
|
| 34 |
+
from __future__ import annotations
|
| 35 |
+
|
| 36 |
+
from dataclasses import dataclass, field
|
| 37 |
+
from typing import Any, Callable, Dict, List, Optional
|
| 38 |
+
|
| 39 |
+
import torch
|
| 40 |
+
import torch.nn as nn
|
| 41 |
+
|
| 42 |
+
from .mutation import MutationPressureTraining, MPTConfig
|
| 43 |
+
from .genome import CodeGenomeInitializer, GenomeConfig
|
| 44 |
+
from .speciation import SpeciationCurriculum, SpeciationConfig, CurriculumPhase
|
| 45 |
+
from .compression import RecursiveSelfCompression, RSCConfig
|
| 46 |
+
from .context_expansion import ContextExpansionProtocol, CEPConfig
|
| 47 |
+
from .adaptive_routing import ADRConfig
|
| 48 |
+
|
| 49 |
+
|
| 50 |
+
@dataclass
|
| 51 |
+
class CyberForgeConfig:
|
| 52 |
+
"""Cấu hình tổng hợp CyberForge training."""
|
| 53 |
+
# Component configs
|
| 54 |
+
genome: GenomeConfig = field(default_factory=GenomeConfig)
|
| 55 |
+
speciation: SpeciationConfig = field(default_factory=SpeciationConfig)
|
| 56 |
+
mpt: MPTConfig = field(default_factory=MPTConfig)
|
| 57 |
+
rsc: RSCConfig = field(default_factory=RSCConfig)
|
| 58 |
+
cep: CEPConfig = field(default_factory=CEPConfig)
|
| 59 |
+
adr: ADRConfig = field(default_factory=ADRConfig)
|
| 60 |
+
|
| 61 |
+
# Total schedule
|
| 62 |
+
total_steps: int = 100_000
|
| 63 |
+
warmup_steps: int = 1_000
|
| 64 |
+
# Phase ratios (override speciation defaults nếu cần)
|
| 65 |
+
speciation_ratio: float = 0.30
|
| 66 |
+
hybridization_ratio: float = 0.30
|
| 67 |
+
generalization_ratio: float = 0.40
|
| 68 |
+
|
| 69 |
+
# Hardware
|
| 70 |
+
use_amp: bool = True
|
| 71 |
+
use_deepspeed: bool = False
|
| 72 |
+
gradient_clip: float = 1.0
|
| 73 |
+
|
| 74 |
+
# Checkpoint
|
| 75 |
+
checkpoint_dir: str = "./checkpoints"
|
| 76 |
+
checkpoint_period: int = 5_000
|
| 77 |
+
log_period: int = 100
|
| 78 |
+
|
| 79 |
+
|
| 80 |
+
class CyberForgeTrainer:
|
| 81 |
+
"""Orchestrator cho toàn bộ CyberGym training.
|
| 82 |
+
|
| 83 |
+
Lưu ý: Trainer này KHÔNG chạy trong môi trường sandbox.
|
| 84 |
+
Nó define toàn bộ pipeline dưới dạng code, để user chạy trên cluster riêng.
|
| 85 |
+
"""
|
| 86 |
+
|
| 87 |
+
def __init__(
|
| 88 |
+
self,
|
| 89 |
+
model: nn.Module,
|
| 90 |
+
config: Optional[CyberForgeConfig] = None,
|
| 91 |
+
train_loader: Optional[Any] = None,
|
| 92 |
+
val_loader: Optional[Any] = None,
|
| 93 |
+
val_loss_fn: Optional[Callable[[nn.Module], float]] = None,
|
| 94 |
+
):
|
| 95 |
+
self.model = model
|
| 96 |
+
self.config = config or CyberForgeConfig()
|
| 97 |
+
self.train_loader = train_loader
|
| 98 |
+
self.val_loader = val_loader
|
| 99 |
+
self.val_loss_fn = val_loss_fn
|
| 100 |
+
|
| 101 |
+
# Sub-components
|
| 102 |
+
self.genome = CodeGenomeInitializer(self.config.genome)
|
| 103 |
+
self.speciation = SpeciationCurriculum(
|
| 104 |
+
self.config.speciation,
|
| 105 |
+
total_steps=self.config.total_steps,
|
| 106 |
+
)
|
| 107 |
+
self.mpt = MutationPressureTraining(
|
| 108 |
+
model,
|
| 109 |
+
config=self.config.mpt,
|
| 110 |
+
val_loss_fn=val_loss_fn,
|
| 111 |
+
)
|
| 112 |
+
self.rsc = RecursiveSelfCompression(model, config=self.config.rsc)
|
| 113 |
+
self.cep = ContextExpansionProtocol(self.config.cep)
|
| 114 |
+
|
| 115 |
+
# Stats
|
| 116 |
+
self._step = 0
|
| 117 |
+
self._stage_stats: List[Dict[str, Any]] = []
|
| 118 |
+
|
| 119 |
+
# ------------------------------------------------------------------
|
| 120 |
+
# Stage 0: Genome Initialization
|
| 121 |
+
# ------------------------------------------------------------------
|
| 122 |
+
|
| 123 |
+
def stage_genome_init(self) -> Dict[str, int]:
|
| 124 |
+
"""Stage 0: Apply Code Genome Init to model weights."""
|
| 125 |
+
stats = self.genome.apply_to(self.model)
|
| 126 |
+
self._stage_stats.append({"stage": "genome_init", **stats})
|
| 127 |
+
return stats
|
| 128 |
+
|
| 129 |
+
# ------------------------------------------------------------------
|
| 130 |
+
# CEP: Apply stage-th context expansion
|
| 131 |
+
# ------------------------------------------------------------------
|
| 132 |
+
|
| 133 |
+
def apply_cep_stage(self, stage_idx: int) -> Dict[str, Any]:
|
| 134 |
+
"""Apply CEP stage-th vào model config."""
|
| 135 |
+
schedule = self.cep.get_schedule()
|
| 136 |
+
if stage_idx < 0 or stage_idx >= len(schedule):
|
| 137 |
+
return {"error": "invalid stage_idx"}
|
| 138 |
+
stage = schedule[stage_idx]
|
| 139 |
+
self.cep.apply_stage_to_config(self.model.config, stage_idx)
|
| 140 |
+
return stage
|
| 141 |
+
|
| 142 |
+
# ------------------------------------------------------------------
|
| 143 |
+
# Step
|
| 144 |
+
# ------------------------------------------------------------------
|
| 145 |
+
|
| 146 |
+
def train_step(self, batch: Any) -> Dict[str, Any]:
|
| 147 |
+
"""One training step — orchestrates all CyberGym components.
|
| 148 |
+
|
| 149 |
+
Args:
|
| 150 |
+
batch: dict with input_ids, attention_mask, labels, (optional) domain
|
| 151 |
+
Returns:
|
| 152 |
+
dict with loss, phase, mpt_stats, rsc_stats, cep_stage
|
| 153 |
+
"""
|
| 154 |
+
if self.train_loader is None and batch is None:
|
| 155 |
+
return {"error": "no batch"}
|
| 156 |
+
|
| 157 |
+
# Determine current phase
|
| 158 |
+
phase = self.speciation.get_phase_at_step(self._step)
|
| 159 |
+
cep_stage = self._cep_stage_for_step(self._step)
|
| 160 |
+
cep_info = self.cep.get_schedule()[cep_stage] if cep_stage < len(self.cep.get_schedule()) else None
|
| 161 |
+
|
| 162 |
+
# Forward pass
|
| 163 |
+
# (Actual forward/backward should be done by caller; here we just dispatch)
|
| 164 |
+
self._step += 1
|
| 165 |
+
|
| 166 |
+
# MPT
|
| 167 |
+
mpt_stats = self.mpt.step()
|
| 168 |
+
|
| 169 |
+
# RSC snapshot
|
| 170 |
+
rsc_snapshot = self.rsc.maybe_snapshot(self._step)
|
| 171 |
+
|
| 172 |
+
return {
|
| 173 |
+
"step": self._step,
|
| 174 |
+
"phase": phase.value,
|
| 175 |
+
"cep_stage": cep_stage,
|
| 176 |
+
"cep_info": cep_info,
|
| 177 |
+
"mpt": mpt_stats,
|
| 178 |
+
"rsc_snapshot_taken": rsc_snapshot,
|
| 179 |
+
}
|
| 180 |
+
|
| 181 |
+
def _cep_stage_for_step(self, step: int) -> int:
|
| 182 |
+
"""Map step → CEP stage."""
|
| 183 |
+
n_stages = len(self.cep.config.stages)
|
| 184 |
+
spec_end = int(self.config.total_steps * self.config.speciation_ratio)
|
| 185 |
+
hyb_end = int(self.config.total_steps * (self.config.speciation_ratio + self.config.hybridization_ratio))
|
| 186 |
+
if step < spec_end:
|
| 187 |
+
return 0 # 32k
|
| 188 |
+
if step < hyb_end:
|
| 189 |
+
progress = (step - spec_end) / max(hyb_end - spec_end, 1)
|
| 190 |
+
return min(n_stages - 1, 1 + int(progress * 2)) # stage 1-2
|
| 191 |
+
progress = (step - hyb_end) / max(self.config.total_steps - hyb_end, 1)
|
| 192 |
+
return min(n_stages - 1, 3 + int(progress * (n_stages - 3))) # stage 3+
|
| 193 |
+
|
| 194 |
+
# ------------------------------------------------------------------
|
| 195 |
+
# Summary
|
| 196 |
+
# ------------------------------------------------------------------
|
| 197 |
+
|
| 198 |
+
def summary(self) -> Dict[str, Any]:
|
| 199 |
+
return {
|
| 200 |
+
"total_steps": self.config.total_steps,
|
| 201 |
+
"phases": {
|
| 202 |
+
"speciation_end": int(self.config.total_steps * self.config.speciation_ratio),
|
| 203 |
+
"hybridization_end": int(self.config.total_steps * (self.config.speciation_ratio + self.config.hybridization_ratio)),
|
| 204 |
+
},
|
| 205 |
+
"genome": self.genome.get_genome_summary(),
|
| 206 |
+
"speciation": self.speciation.summary(),
|
| 207 |
+
"cep": self.cep.summary(),
|
| 208 |
+
"mpt_stats": self.mpt.stats(),
|
| 209 |
+
"rsc_stats": self.rsc.stats(),
|
| 210 |
+
"adr": {
|
| 211 |
+
"min_active": self.config.adr.min_active_experts,
|
| 212 |
+
"max_active": self.config.adr.max_active_experts,
|
| 213 |
+
},
|
| 214 |
+
"stage_history": self._stage_stats,
|
| 215 |
+
}
|
| 216 |
+
|
| 217 |
+
def print_summary(self) -> None:
|
| 218 |
+
"""In tóm tắt pipeline."""
|
| 219 |
+
s = self.summary()
|
| 220 |
+
print("=" * 72)
|
| 221 |
+
print(" CyberForge Training Pipeline Summary")
|
| 222 |
+
print("=" * 72)
|
| 223 |
+
print(f" Total steps: {s['total_steps']:,}")
|
| 224 |
+
print(f" Speciation phase end: {s['phases']['speciation_end']:,}")
|
| 225 |
+
print(f" Hybridization end: {s['phases']['hybridization_end']:,}")
|
| 226 |
+
print("-" * 72)
|
| 227 |
+
print(f" Genome motifs: {s['genome']['num_motifs']}")
|
| 228 |
+
print(f" Genome inject layers: {s['genome']['injection_layers']}")
|
| 229 |
+
print("-" * 72)
|
| 230 |
+
print(f" CEP stages: {len(s['cep']['stages'])}")
|
| 231 |
+
print(f" CEP growth: {s['cep']['total_context_growth']}")
|
| 232 |
+
print(f" CEP growth factor: {s['cep']['growth_factor']:.0f}x")
|
| 233 |
+
print("-" * 72)
|
| 234 |
+
print(f" ADR active experts: {s['adr']['min_active']}..{s['adr']['max_active']}")
|
| 235 |
+
print(f" MPT acceptance rate: {s['mpt_stats'].get('acceptance_rate', 0):.1%}")
|
| 236 |
+
print(f" RSC snapshots: {s['rsc_stats'].get('snapshots_taken', 0)}")
|
| 237 |
+
print("=" * 72)
|
nexus/data/__init__.py
ADDED
|
@@ -0,0 +1,42 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
Nexus Data Module - v0.2 NEW
|
| 3 |
+
============================
|
| 4 |
+
Pipeline thu thập và xử lý training data.
|
| 5 |
+
|
| 6 |
+
Sources:
|
| 7 |
+
- GitHubCollector: Code từ public GitHub repos
|
| 8 |
+
- HuggingFaceCollector: Datasets từ HuggingFace Hub
|
| 9 |
+
- ArxivCollector: Scientific papers
|
| 10 |
+
- WikipediaCollector: General knowledge
|
| 11 |
+
- StackOverflowCollector: Q&A pairs
|
| 12 |
+
|
| 13 |
+
Processors:
|
| 14 |
+
- TextCleaner: Làm sạch text
|
| 15 |
+
- CodeFormatter: Format code samples
|
| 16 |
+
- Deduplicator: Loại bỏ duplicates (MinHash)
|
| 17 |
+
- QualityFilter: Lọc low-quality samples
|
| 18 |
+
"""
|
| 19 |
+
|
| 20 |
+
from .collectors.github_collector import GitHubCollector
|
| 21 |
+
from .collectors.huggingface_collector import HuggingFaceCollector
|
| 22 |
+
from .collectors.arxiv_collector import ArxivCollector
|
| 23 |
+
from .collectors.wikipedia_collector import WikipediaCollector
|
| 24 |
+
from .collectors.stackoverflow_collector import StackOverflowCollector
|
| 25 |
+
from .processors.cleaner import TextCleaner
|
| 26 |
+
from .processors.deduplicator import Deduplicator
|
| 27 |
+
from .processors.quality_filter import QualityFilter
|
| 28 |
+
from .processors.code_formatter import CodeFormatter
|
| 29 |
+
from .curriculum import CurriculumLearning
|
| 30 |
+
|
| 31 |
+
__all__ = [
|
| 32 |
+
"GitHubCollector",
|
| 33 |
+
"HuggingFaceCollector",
|
| 34 |
+
"ArxivCollector",
|
| 35 |
+
"WikipediaCollector",
|
| 36 |
+
"StackOverflowCollector",
|
| 37 |
+
"TextCleaner",
|
| 38 |
+
"Deduplicator",
|
| 39 |
+
"QualityFilter",
|
| 40 |
+
"CodeFormatter",
|
| 41 |
+
"CurriculumLearning",
|
| 42 |
+
]
|
nexus/data/collectors/__init__.py
ADDED
|
@@ -0,0 +1,39 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Data collectors package (v0.3 expanded).
|
| 2 |
+
|
| 3 |
+
v0.2: GitHub, HuggingFace, arXiv, Wikipedia, StackOverflow
|
| 4 |
+
v0.3: + The-Stack, StarCoder2-data, Python-Alpaca
|
| 5 |
+
"""
|
| 6 |
+
from .github_collector import GitHubCollector
|
| 7 |
+
from .huggingface_collector import HuggingFaceCollector
|
| 8 |
+
from .arxiv_collector import ArxivCollector
|
| 9 |
+
from .wikipedia_collector import WikipediaCollector
|
| 10 |
+
from .stackoverflow_collector import StackOverflowCollector
|
| 11 |
+
|
| 12 |
+
# v0.3 NEW
|
| 13 |
+
try:
|
| 14 |
+
from .the_stack_collector import TheStackCollector
|
| 15 |
+
except ImportError:
|
| 16 |
+
TheStackCollector = None # type: ignore
|
| 17 |
+
|
| 18 |
+
try:
|
| 19 |
+
from .starcoder2_collector import StarCoder2Collector
|
| 20 |
+
except ImportError:
|
| 21 |
+
StarCoder2Collector = None # type: ignore
|
| 22 |
+
|
| 23 |
+
try:
|
| 24 |
+
from .python_alpaca_collector import PythonAlpacaCollector
|
| 25 |
+
except ImportError:
|
| 26 |
+
PythonAlpacaCollector = None # type: ignore
|
| 27 |
+
|
| 28 |
+
|
| 29 |
+
__all__ = [
|
| 30 |
+
"GitHubCollector",
|
| 31 |
+
"HuggingFaceCollector",
|
| 32 |
+
"ArxivCollector",
|
| 33 |
+
"WikipediaCollector",
|
| 34 |
+
"StackOverflowCollector",
|
| 35 |
+
# v0.3 NEW
|
| 36 |
+
"TheStackCollector",
|
| 37 |
+
"StarCoder2Collector",
|
| 38 |
+
"PythonAlpacaCollector",
|
| 39 |
+
]
|
nexus/data/collectors/arxiv_collector.py
ADDED
|
@@ -0,0 +1,225 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
Arxiv Collector - Thu thập scientific papers từ arXiv
|
| 3 |
+
======================================================
|
| 4 |
+
"""
|
| 5 |
+
from __future__ import annotations
|
| 6 |
+
|
| 7 |
+
import os
|
| 8 |
+
import logging
|
| 9 |
+
import urllib.request
|
| 10 |
+
import xml.etree.ElementTree as ET
|
| 11 |
+
from typing import List, Dict, Optional, Iterator, Any
|
| 12 |
+
from dataclasses import dataclass, field
|
| 13 |
+
import time
|
| 14 |
+
|
| 15 |
+
logger = logging.getLogger(__name__)
|
| 16 |
+
|
| 17 |
+
|
| 18 |
+
@dataclass
|
| 19 |
+
class ArxivPaper:
|
| 20 |
+
"""Thông tin một arXiv paper."""
|
| 21 |
+
arxiv_id: str
|
| 22 |
+
title: str
|
| 23 |
+
authors: List[str]
|
| 24 |
+
abstract: str
|
| 25 |
+
categories: List[str]
|
| 26 |
+
published: str
|
| 27 |
+
pdf_url: str
|
| 28 |
+
|
| 29 |
+
|
| 30 |
+
class ArxivCollector:
|
| 31 |
+
"""Collect papers từ arXiv API.
|
| 32 |
+
|
| 33 |
+
Usage:
|
| 34 |
+
collector = ArxivCollector()
|
| 35 |
+
papers = collector.search("transformer attention", max_results=100)
|
| 36 |
+
for paper in papers:
|
| 37 |
+
print(paper.title)
|
| 38 |
+
"""
|
| 39 |
+
|
| 40 |
+
BASE_URL = "http://export.arxiv.org/api/query"
|
| 41 |
+
|
| 42 |
+
CATEGORIES = [
|
| 43 |
+
"cs.CL", # Computation and Language (NLP)
|
| 44 |
+
"cs.LG", # Machine Learning
|
| 45 |
+
"cs.AI", # Artificial Intelligence
|
| 46 |
+
"cs.SE", # Software Engineering
|
| 47 |
+
"cs.PL", # Programming Languages
|
| 48 |
+
"cs.CV", # Computer Vision
|
| 49 |
+
"stat.ML", # Statistics - Machine Learning
|
| 50 |
+
]
|
| 51 |
+
|
| 52 |
+
def __init__(self, delay: float = 3.0):
|
| 53 |
+
"""Args:
|
| 54 |
+
delay: Seconds between API calls (arXiv rate limit: 1 req per 3s)
|
| 55 |
+
"""
|
| 56 |
+
self.delay = delay
|
| 57 |
+
self._last_request = 0.0
|
| 58 |
+
|
| 59 |
+
def search(
|
| 60 |
+
self,
|
| 61 |
+
query: str,
|
| 62 |
+
max_results: int = 100,
|
| 63 |
+
category: Optional[str] = None,
|
| 64 |
+
sort_by: str = "relevance",
|
| 65 |
+
) -> List[ArxivPaper]:
|
| 66 |
+
"""Search arXiv papers.
|
| 67 |
+
|
| 68 |
+
Args:
|
| 69 |
+
query: Search query
|
| 70 |
+
max_results: Max papers to return
|
| 71 |
+
category: Filter by arXiv category (e.g. "cs.CL")
|
| 72 |
+
sort_by: "relevance", "lastUpdatedDate", "submittedDate"
|
| 73 |
+
"""
|
| 74 |
+
self._rate_limit()
|
| 75 |
+
|
| 76 |
+
params = {
|
| 77 |
+
"search_query": self._build_query(query, category),
|
| 78 |
+
"start": 0,
|
| 79 |
+
"max_results": min(max_results, 2000),
|
| 80 |
+
"sortBy": sort_by,
|
| 81 |
+
"sortOrder": "descending",
|
| 82 |
+
}
|
| 83 |
+
|
| 84 |
+
url = f"{self.BASE_URL}?{'&'.join(f'{k}={v}' for k, v in params.items())}"
|
| 85 |
+
|
| 86 |
+
try:
|
| 87 |
+
req = urllib.request.Request(url, headers={"User-Agent": "NexusCoder-Collector/0.2"})
|
| 88 |
+
with urllib.request.urlopen(req, timeout=30) as response:
|
| 89 |
+
xml_data = response.read().decode()
|
| 90 |
+
|
| 91 |
+
return self._parse_response(xml_data)
|
| 92 |
+
except Exception as e:
|
| 93 |
+
logger.error(f"arXiv search failed: {e}")
|
| 94 |
+
return []
|
| 95 |
+
|
| 96 |
+
def _build_query(self, query: str, category: Optional[str]) -> str:
|
| 97 |
+
"""Build arXiv query string (URL-encoded for safety)."""
|
| 98 |
+
# v0.4 fix: use urllib.parse.quote so special chars in query don't break URL.
|
| 99 |
+
import urllib.parse
|
| 100 |
+
parts = []
|
| 101 |
+
if query:
|
| 102 |
+
q = urllib.parse.quote(query, safe='')
|
| 103 |
+
parts.append(f'(abs:"{q}" OR ti:"{q}")')
|
| 104 |
+
if category:
|
| 105 |
+
parts.append(f"cat:{category}")
|
| 106 |
+
return " AND ".join(parts) if parts else "all:*"
|
| 107 |
+
|
| 108 |
+
def _parse_response(self, xml_data: str) -> List[ArxivPaper]:
|
| 109 |
+
"""Parse arXiv API XML response."""
|
| 110 |
+
ns = {
|
| 111 |
+
"atom": "http://www.w3.org/2005/Atom",
|
| 112 |
+
"arxiv": "http://arxiv.org/schemas/atom",
|
| 113 |
+
}
|
| 114 |
+
|
| 115 |
+
papers = []
|
| 116 |
+
try:
|
| 117 |
+
root = ET.fromstring(xml_data)
|
| 118 |
+
for entry in root.findall("atom:entry", ns):
|
| 119 |
+
# v0.4 fix: None-safe access for each field
|
| 120 |
+
id_el = entry.find("atom:id", ns)
|
| 121 |
+
arxiv_id = (
|
| 122 |
+
id_el.text.split("/")[-1]
|
| 123 |
+
if id_el is not None and id_el.text
|
| 124 |
+
else ""
|
| 125 |
+
)
|
| 126 |
+
|
| 127 |
+
title_el = entry.find("atom:title", ns)
|
| 128 |
+
title = (
|
| 129 |
+
title_el.text.strip().replace("\n", " ")
|
| 130 |
+
if title_el is not None and title_el.text
|
| 131 |
+
else ""
|
| 132 |
+
)
|
| 133 |
+
|
| 134 |
+
summary_el = entry.find("atom:summary", ns)
|
| 135 |
+
abstract = (
|
| 136 |
+
summary_el.text.strip().replace("\n", " ")
|
| 137 |
+
if summary_el is not None and summary_el.text
|
| 138 |
+
else ""
|
| 139 |
+
)
|
| 140 |
+
|
| 141 |
+
published_el = entry.find("atom:published", ns)
|
| 142 |
+
published = (
|
| 143 |
+
published_el.text
|
| 144 |
+
if published_el is not None and published_el.text
|
| 145 |
+
else ""
|
| 146 |
+
)
|
| 147 |
+
|
| 148 |
+
authors = []
|
| 149 |
+
for author in entry.findall("atom:author", ns):
|
| 150 |
+
name = author.find("atom:name", ns)
|
| 151 |
+
if name is not None:
|
| 152 |
+
authors.append(name.text)
|
| 153 |
+
|
| 154 |
+
categories = []
|
| 155 |
+
for link in entry.findall("atom:link", ns):
|
| 156 |
+
if link.get("title") == "pdf":
|
| 157 |
+
pdf_url = link.get("href")
|
| 158 |
+
|
| 159 |
+
# Get categories
|
| 160 |
+
for cat in entry.findall("atom:category", ns):
|
| 161 |
+
term = cat.get("term")
|
| 162 |
+
if term:
|
| 163 |
+
categories.append(term)
|
| 164 |
+
|
| 165 |
+
papers.append(ArxivPaper(
|
| 166 |
+
arxiv_id=arxiv_id,
|
| 167 |
+
title=title,
|
| 168 |
+
authors=authors,
|
| 169 |
+
abstract=abstract,
|
| 170 |
+
categories=categories,
|
| 171 |
+
published=published,
|
| 172 |
+
pdf_url=f"https://arxiv.org/pdf/{arxiv_id}",
|
| 173 |
+
))
|
| 174 |
+
except Exception as e:
|
| 175 |
+
logger.error(f"Parse error: {e}")
|
| 176 |
+
|
| 177 |
+
return papers
|
| 178 |
+
|
| 179 |
+
def _rate_limit(self) -> None:
|
| 180 |
+
"""Enforce rate limit."""
|
| 181 |
+
elapsed = time.time() - self._last_request
|
| 182 |
+
if elapsed < self.delay:
|
| 183 |
+
time.sleep(self.delay - elapsed)
|
| 184 |
+
self._last_request = time.time()
|
| 185 |
+
|
| 186 |
+
def collect(self, queries: List[str], max_per_query: int = 100) -> Iterator[Dict[str, Any]]:
|
| 187 |
+
"""Collect papers from multiple queries, yield as text samples."""
|
| 188 |
+
for query in queries:
|
| 189 |
+
papers = self.search(query, max_results=max_per_query)
|
| 190 |
+
for paper in papers:
|
| 191 |
+
yield {
|
| 192 |
+
"text": f"Title: {paper.title}\n\nAuthors: {', '.join(paper.authors)}\n\nAbstract: {paper.abstract}",
|
| 193 |
+
"source": "arxiv",
|
| 194 |
+
"language": "en",
|
| 195 |
+
"metadata": {
|
| 196 |
+
"arxiv_id": paper.arxiv_id,
|
| 197 |
+
"categories": paper.categories,
|
| 198 |
+
"published": paper.published,
|
| 199 |
+
},
|
| 200 |
+
}
|
| 201 |
+
|
| 202 |
+
|
| 203 |
+
# Curated search queries for ML/CS topics
|
| 204 |
+
CURATED_QUERIES = [
|
| 205 |
+
"transformer architecture",
|
| 206 |
+
"mixture of experts",
|
| 207 |
+
"large language model",
|
| 208 |
+
"attention mechanism",
|
| 209 |
+
"code generation",
|
| 210 |
+
"program synthesis",
|
| 211 |
+
"neural machine translation",
|
| 212 |
+
"retrieval augmented generation",
|
| 213 |
+
"instruction tuning",
|
| 214 |
+
"reinforcement learning human feedback",
|
| 215 |
+
"chain of thought reasoning",
|
| 216 |
+
"prompt engineering",
|
| 217 |
+
"fine-tuning language model",
|
| 218 |
+
"quantization neural network",
|
| 219 |
+
"knowledge distillation",
|
| 220 |
+
"multi-agent systems",
|
| 221 |
+
"tool use language model",
|
| 222 |
+
"code completion",
|
| 223 |
+
"static analysis",
|
| 224 |
+
"program verification",
|
| 225 |
+
]
|
nexus/data/collectors/github_collector.py
ADDED
|
@@ -0,0 +1,420 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
GitHub Collector - Thu thập code từ GitHub repositories
|
| 3 |
+
========================================================
|
| 4 |
+
thu thập dữ liệu training từ public GitHub repos.
|
| 5 |
+
|
| 6 |
+
Features:
|
| 7 |
+
- Clone & extract code từ repos
|
| 8 |
+
- Filter theo language, file size, license
|
| 9 |
+
- Extract functions, classes, docstrings
|
| 10 |
+
- Rate limit aware (GitHub API: 5000 req/h với token)
|
| 11 |
+
- Parallel fetching
|
| 12 |
+
"""
|
| 13 |
+
from __future__ import annotations
|
| 14 |
+
|
| 15 |
+
import os
|
| 16 |
+
import subprocess
|
| 17 |
+
import tempfile
|
| 18 |
+
import logging
|
| 19 |
+
from typing import List, Dict, Optional, Iterator, Tuple
|
| 20 |
+
from dataclasses import dataclass, field
|
| 21 |
+
from pathlib import Path
|
| 22 |
+
import json
|
| 23 |
+
import time
|
| 24 |
+
|
| 25 |
+
logger = logging.getLogger(__name__)
|
| 26 |
+
|
| 27 |
+
|
| 28 |
+
@dataclass
|
| 29 |
+
class GitHubRepo:
|
| 30 |
+
"""Thông tin một GitHub repo để collect."""
|
| 31 |
+
owner: str
|
| 32 |
+
name: str
|
| 33 |
+
branch: str = "main"
|
| 34 |
+
languages: List[str] = field(default_factory=lambda: ["python"])
|
| 35 |
+
max_files: int = 1000
|
| 36 |
+
max_file_size_kb: int = 100
|
| 37 |
+
license_filter: List[str] = field(default_factory=lambda: ["MIT", "Apache-2.0", "BSD", "GPL"])
|
| 38 |
+
|
| 39 |
+
@property
|
| 40 |
+
def url(self) -> str:
|
| 41 |
+
return f"https://github.com/{self.owner}/{self.name}.git"
|
| 42 |
+
|
| 43 |
+
@property
|
| 44 |
+
def api_url(self) -> str:
|
| 45 |
+
return f"https://api.github.com/repos/{self.owner}/{self.name}"
|
| 46 |
+
|
| 47 |
+
|
| 48 |
+
@dataclass
|
| 49 |
+
class CodeSample:
|
| 50 |
+
"""Một sample code được thu thập."""
|
| 51 |
+
repo: str
|
| 52 |
+
file_path: str
|
| 53 |
+
language: str
|
| 54 |
+
content: str
|
| 55 |
+
size: int
|
| 56 |
+
license: Optional[str] = None
|
| 57 |
+
quality_score: float = 0.0
|
| 58 |
+
|
| 59 |
+
|
| 60 |
+
class GitHubCollector:
|
| 61 |
+
"""Collect training data từ GitHub repositories.
|
| 62 |
+
|
| 63 |
+
Usage:
|
| 64 |
+
collector = GitHubCollector(token="ghp_xxx")
|
| 65 |
+
repos = [
|
| 66 |
+
GitHubRepo("python", "cpython", languages=["python"]),
|
| 67 |
+
GitHubRepo("pallets", "flask"),
|
| 68 |
+
]
|
| 69 |
+
for sample in collector.collect(repos):
|
| 70 |
+
print(sample.file_path, len(sample.content))
|
| 71 |
+
"""
|
| 72 |
+
|
| 73 |
+
EXTENSIONS = {
|
| 74 |
+
"python": [".py"],
|
| 75 |
+
"javascript": [".js", ".mjs", ".jsx"],
|
| 76 |
+
"typescript": [".ts", ".tsx"],
|
| 77 |
+
"go": [".go"],
|
| 78 |
+
"rust": [".rs"],
|
| 79 |
+
"java": [".java"],
|
| 80 |
+
"c": [".c", ".h"],
|
| 81 |
+
"cpp": [".cpp", ".cc", ".cxx", ".hpp", ".hxx"],
|
| 82 |
+
"csharp": [".cs"],
|
| 83 |
+
"ruby": [".rb"],
|
| 84 |
+
"php": [".php"],
|
| 85 |
+
"swift": [".swift"],
|
| 86 |
+
"kotlin": [".kt"],
|
| 87 |
+
"scala": [".scala"],
|
| 88 |
+
"sql": [".sql"],
|
| 89 |
+
"shell": [".sh", ".bash"],
|
| 90 |
+
"yaml": [".yaml", ".yml"],
|
| 91 |
+
"markdown": [".md", ".markdown"],
|
| 92 |
+
}
|
| 93 |
+
|
| 94 |
+
SKIP_DIRS = {
|
| 95 |
+
"node_modules", "vendor", "venv", ".venv", "env", "__pycache__",
|
| 96 |
+
".git", ".github", "dist", "build", "target", "out", "bin",
|
| 97 |
+
".idea", ".vscode", "coverage", ".cache", ".eggs", ".tox",
|
| 98 |
+
}
|
| 99 |
+
|
| 100 |
+
def __init__(
|
| 101 |
+
self,
|
| 102 |
+
token: Optional[str] = None,
|
| 103 |
+
cache_dir: str = "./data_cache/github",
|
| 104 |
+
max_concurrent: int = 4,
|
| 105 |
+
):
|
| 106 |
+
self.token = token or os.environ.get("GITHUB_TOKEN")
|
| 107 |
+
self.cache_dir = cache_dir
|
| 108 |
+
self.max_concurrent = max_concurrent
|
| 109 |
+
os.makedirs(cache_dir, exist_ok=True)
|
| 110 |
+
|
| 111 |
+
def collect(self, repos: List[GitHubRepo]) -> Iterator[CodeSample]:
|
| 112 |
+
"""Collect code samples từ list of repos.
|
| 113 |
+
|
| 114 |
+
Yields:
|
| 115 |
+
CodeSample objects
|
| 116 |
+
"""
|
| 117 |
+
for repo in repos:
|
| 118 |
+
try:
|
| 119 |
+
yield from self._collect_repo(repo)
|
| 120 |
+
except Exception as e:
|
| 121 |
+
logger.error(f"Failed to collect {repo.url}: {e}")
|
| 122 |
+
continue
|
| 123 |
+
|
| 124 |
+
def _collect_repo(self, repo: GitHubRepo) -> Iterator[CodeSample]:
|
| 125 |
+
"""Collect từ một repo."""
|
| 126 |
+
cache_path = os.path.join(self.cache_dir, f"{repo.owner}_{repo.name}")
|
| 127 |
+
|
| 128 |
+
# Clone if not cached
|
| 129 |
+
if not os.path.exists(cache_path):
|
| 130 |
+
logger.info(f"Cloning {repo.url}...")
|
| 131 |
+
try:
|
| 132 |
+
# v0.4 fix: try main, then fall back to master, then default branch.
|
| 133 |
+
# Many older repos use `master` as their default branch.
|
| 134 |
+
clone_ok = False
|
| 135 |
+
last_err = ""
|
| 136 |
+
for branch in (repo.branch, "main", "master"):
|
| 137 |
+
try:
|
| 138 |
+
subprocess.run(
|
| 139 |
+
["git", "clone", "--depth", "1", "--branch", branch, repo.url, cache_path],
|
| 140 |
+
check=True,
|
| 141 |
+
capture_output=True,
|
| 142 |
+
timeout=300,
|
| 143 |
+
)
|
| 144 |
+
clone_ok = True
|
| 145 |
+
break
|
| 146 |
+
except subprocess.CalledProcessError as e:
|
| 147 |
+
last_err = (e.stderr or b"").decode(errors="replace")[:200]
|
| 148 |
+
# Clean up partial clone for next attempt
|
| 149 |
+
if os.path.exists(cache_path):
|
| 150 |
+
import shutil
|
| 151 |
+
shutil.rmtree(cache_path, ignore_errors=True)
|
| 152 |
+
if not clone_ok:
|
| 153 |
+
logger.error(f"Clone failed for {repo.url} (tried main & master): {last_err}")
|
| 154 |
+
return
|
| 155 |
+
except subprocess.TimeoutExpired:
|
| 156 |
+
logger.error(f"Clone timeout for {repo.url}")
|
| 157 |
+
return
|
| 158 |
+
|
| 159 |
+
# Walk and collect files
|
| 160 |
+
count = 0
|
| 161 |
+
for root, dirs, files in os.walk(cache_path):
|
| 162 |
+
# Filter dirs in-place
|
| 163 |
+
dirs[:] = [d for d in dirs if d not in self.SKIP_DIRS and not d.startswith(".")]
|
| 164 |
+
|
| 165 |
+
for fname in files:
|
| 166 |
+
if count >= repo.max_files:
|
| 167 |
+
return
|
| 168 |
+
|
| 169 |
+
ext = os.path.splitext(fname)[1].lower()
|
| 170 |
+
lang = self._detect_language(ext)
|
| 171 |
+
if lang is None or (repo.languages and lang not in repo.languages):
|
| 172 |
+
continue
|
| 173 |
+
|
| 174 |
+
fpath = os.path.join(root, fname)
|
| 175 |
+
|
| 176 |
+
# Size check
|
| 177 |
+
try:
|
| 178 |
+
size = os.path.getsize(fpath)
|
| 179 |
+
if size > repo.max_file_size_kb * 1024 or size < 100:
|
| 180 |
+
continue
|
| 181 |
+
except OSError:
|
| 182 |
+
continue
|
| 183 |
+
|
| 184 |
+
# Read
|
| 185 |
+
try:
|
| 186 |
+
with open(fpath, "r", encoding="utf-8", errors="replace") as f:
|
| 187 |
+
content = f.read()
|
| 188 |
+
except Exception:
|
| 189 |
+
continue
|
| 190 |
+
|
| 191 |
+
# Quality filter
|
| 192 |
+
if not self._is_quality(content, lang):
|
| 193 |
+
continue
|
| 194 |
+
|
| 195 |
+
rel_path = os.path.relpath(fpath, cache_path)
|
| 196 |
+
|
| 197 |
+
yield CodeSample(
|
| 198 |
+
repo=f"{repo.owner}/{repo.name}",
|
| 199 |
+
file_path=rel_path,
|
| 200 |
+
language=lang,
|
| 201 |
+
content=content,
|
| 202 |
+
size=size,
|
| 203 |
+
quality_score=self._score_quality(content, lang),
|
| 204 |
+
)
|
| 205 |
+
count += 1
|
| 206 |
+
|
| 207 |
+
def _detect_language(self, ext: str) -> Optional[str]:
|
| 208 |
+
for lang, exts in self.EXTENSIONS.items():
|
| 209 |
+
if ext in exts:
|
| 210 |
+
return lang
|
| 211 |
+
return None
|
| 212 |
+
|
| 213 |
+
def _is_quality(self, content: str, lang: str) -> bool:
|
| 214 |
+
"""Basic quality filter."""
|
| 215 |
+
if len(content) < 50:
|
| 216 |
+
return False
|
| 217 |
+
if len(content) > 100000: # Skip huge files
|
| 218 |
+
return False
|
| 219 |
+
# Skip if too many non-printable chars
|
| 220 |
+
non_print = sum(1 for c in content if not c.isprintable() and c not in "\n\r\t")
|
| 221 |
+
if non_print / len(content) > 0.05:
|
| 222 |
+
return False
|
| 223 |
+
# Skip auto-generated files
|
| 224 |
+
if "auto-generated" in content[:200].lower():
|
| 225 |
+
return False
|
| 226 |
+
if "DO NOT EDIT" in content[:200]:
|
| 227 |
+
return False
|
| 228 |
+
return True
|
| 229 |
+
|
| 230 |
+
def _score_quality(self, content: str, lang: str) -> float:
|
| 231 |
+
"""Score quality [0.0, 1.0]."""
|
| 232 |
+
score = 0.5
|
| 233 |
+
# Has docstrings/comments
|
| 234 |
+
if lang == "python":
|
| 235 |
+
if '"""' in content or "'''" in content:
|
| 236 |
+
score += 0.2
|
| 237 |
+
if "# " in content:
|
| 238 |
+
score += 0.1
|
| 239 |
+
# Has type hints
|
| 240 |
+
if "->" in content or ": int" in content or ": str" in content:
|
| 241 |
+
score += 0.1
|
| 242 |
+
# Reasonable length
|
| 243 |
+
lines = content.count("\n")
|
| 244 |
+
if 20 <= lines <= 500:
|
| 245 |
+
score += 0.1
|
| 246 |
+
return min(1.0, score)
|
| 247 |
+
|
| 248 |
+
def search_repos(
|
| 249 |
+
self,
|
| 250 |
+
query: str,
|
| 251 |
+
language: str = "python",
|
| 252 |
+
sort: str = "stars",
|
| 253 |
+
max_results: int = 50,
|
| 254 |
+
) -> List[GitHubRepo]:
|
| 255 |
+
"""Search GitHub repos by query (requires token)."""
|
| 256 |
+
if not self.token:
|
| 257 |
+
logger.warning("No GitHub token - cannot search")
|
| 258 |
+
return []
|
| 259 |
+
|
| 260 |
+
import urllib.request
|
| 261 |
+
import urllib.parse
|
| 262 |
+
|
| 263 |
+
params = urllib.parse.urlencode({
|
| 264 |
+
"q": f"{query} language:{language}",
|
| 265 |
+
"sort": sort,
|
| 266 |
+
"order": "desc",
|
| 267 |
+
"per_page": min(max_results, 100),
|
| 268 |
+
})
|
| 269 |
+
url = f"https://api.github.com/search/repositories?{params}"
|
| 270 |
+
|
| 271 |
+
req = urllib.request.Request(url, headers={
|
| 272 |
+
"Authorization": f"token {self.token}",
|
| 273 |
+
"Accept": "application/vnd.github.v3+json",
|
| 274 |
+
"User-Agent": "NexusCoder-DataCollector/0.2",
|
| 275 |
+
})
|
| 276 |
+
|
| 277 |
+
try:
|
| 278 |
+
with urllib.request.urlopen(req, timeout=30) as response:
|
| 279 |
+
data = json.loads(response.read().decode())
|
| 280 |
+
|
| 281 |
+
repos = []
|
| 282 |
+
for item in data.get("items", [])[:max_results]:
|
| 283 |
+
repos.append(GitHubRepo(
|
| 284 |
+
owner=item["owner"]["login"],
|
| 285 |
+
name=item["name"],
|
| 286 |
+
languages=[language],
|
| 287 |
+
))
|
| 288 |
+
return repos
|
| 289 |
+
except Exception as e:
|
| 290 |
+
logger.error(f"GitHub search failed: {e}")
|
| 291 |
+
return []
|
| 292 |
+
|
| 293 |
+
|
| 294 |
+
# =============================================================================
|
| 295 |
+
# Curated list of high-quality repos for training
|
| 296 |
+
# =============================================================================
|
| 297 |
+
|
| 298 |
+
CURATED_REPOS: List[GitHubRepo] = [
|
| 299 |
+
# Python core
|
| 300 |
+
GitHubRepo("python", "cpython", languages=["python"], max_files=2000),
|
| 301 |
+
GitHubRepo("pallets", "flask", languages=["python"]),
|
| 302 |
+
GitHubRepo("pallets", "django", languages=["python"], max_files=2000),
|
| 303 |
+
GitHubRepo("pallets", "click", languages=["python"]),
|
| 304 |
+
GitHubRepo("psf", "requests", languages=["python"]),
|
| 305 |
+
GitHubRepo("psf", "requests-html", languages=["python"]),
|
| 306 |
+
|
| 307 |
+
# Data science
|
| 308 |
+
GitHubRepo("numpy", "numpy", languages=["python"], max_files=2000),
|
| 309 |
+
GitHubRepo("pandas-dev", "pandas", languages=["python"], max_files=2000),
|
| 310 |
+
GitHubRepo("scipy", "scipy", languages=["python"], max_files=2000),
|
| 311 |
+
GitHubRepo("matplotlib", "matplotlib", languages=["python"], max_files=2000),
|
| 312 |
+
GitHubRepo("scikit-learn", "scikit-learn", languages=["python"], max_files=2000),
|
| 313 |
+
|
| 314 |
+
# ML/DL
|
| 315 |
+
GitHubRepo("pytorch", "pytorch", languages=["python", "cpp"], max_files=2000),
|
| 316 |
+
GitHubRepo("tensorflow", "tensorflow", languages=["python", "cpp"], max_files=2000),
|
| 317 |
+
GitHubRepo("huggingface", "transformers", languages=["python"], max_files=2000),
|
| 318 |
+
GitHubRepo("huggingface", "datasets", languages=["python"]),
|
| 319 |
+
GitHubRepo("huggingface", "tokenizers", languages=["python", "rust"]),
|
| 320 |
+
GitHubRepo("langchain-ai", "langchain", languages=["python"], max_files=2000),
|
| 321 |
+
GitHubRepo("ollama", "ollama-python", languages=["python"]),
|
| 322 |
+
|
| 323 |
+
# Web frameworks
|
| 324 |
+
GitHubRepo("tiangolo", "fastapi", languages=["python"], max_files=2000),
|
| 325 |
+
GitHubRepo("encode", "starlette", languages=["python"]),
|
| 326 |
+
GitHubRepo("encode", "uvicorn", languages=["python"]),
|
| 327 |
+
GitHubRepo("tornadoweb", "tornado", languages=["python"]),
|
| 328 |
+
GitHubRepo("Sanic", "sanic", languages=["python"]),
|
| 329 |
+
|
| 330 |
+
# CLI
|
| 331 |
+
GitHubRepo("click", "click", languages=["python"]),
|
| 332 |
+
GitHubRepo("prompt-toolkit", "python-prompt-toolkit", languages=["python"]),
|
| 333 |
+
GitHubRepo("Textualize", "rich", languages=["python"]),
|
| 334 |
+
GitHubRepo("Textualize", "textual", languages=["python"]),
|
| 335 |
+
|
| 336 |
+
# Tools
|
| 337 |
+
GitHubRepo("pytest-dev", "pytest", languages=["python"]),
|
| 338 |
+
GitHubRepo("pypa", "pip", languages=["python"]),
|
| 339 |
+
GitHubRepo("pypa", "setuptools", languages=["python"]),
|
| 340 |
+
GitHubRepo("mkdocs", "mkdocs", languages=["python"]),
|
| 341 |
+
GitHubRepo("sphinx-doc", "sphinx", languages=["python"]),
|
| 342 |
+
|
| 343 |
+
# Async
|
| 344 |
+
GitHubRepo("MagicStack", "uvloop", languages=["python", "c"]),
|
| 345 |
+
GitHubRepo("aio-libs", "aiohttp", languages=["python"], max_files=2000),
|
| 346 |
+
GitHubRepo("aio-libs", "aiomysql", languages=["python"]),
|
| 347 |
+
GitHubRepo("aio-libs", "aiopg", languages=["python"]),
|
| 348 |
+
|
| 349 |
+
# Database
|
| 350 |
+
GitHubRepo("sqlalchemy", "sqlalchemy", languages=["python"], max_files=2000),
|
| 351 |
+
GitHubRepo("mongodb", "mongo-python-driver", languages=["python"]),
|
| 352 |
+
GitHubRepo("redis", "redis-py", languages=["python"]),
|
| 353 |
+
GitHubRepo("coleifer", "peewee", languages=["python"]),
|
| 354 |
+
|
| 355 |
+
# Other useful
|
| 356 |
+
GitHubRepo("psf", "black", languages=["python"]),
|
| 357 |
+
GitHubRepo("pycqa", "flake8", languages=["python"]),
|
| 358 |
+
GitHubRepo("pycqa", "isort", languages=["python"]),
|
| 359 |
+
GitHubRepo("python-attrs", "attrs", languages=["python"]),
|
| 360 |
+
GitHubRepo("pydantic", "pydantic", languages=["python"]),
|
| 361 |
+
GitHubRepo("encode", "httpx", languages=["python"]),
|
| 362 |
+
GitHubRepo("httpie", "httpie", languages=["python"]),
|
| 363 |
+
GitHubRepo("pypa", "virtualenv", languages=["python"]),
|
| 364 |
+
GitHubRepo("pypa", "build", languages=["python"]),
|
| 365 |
+
|
| 366 |
+
# JavaScript/TypeScript
|
| 367 |
+
GitHubRepo("facebook", "react", languages=["javascript", "typescript"], max_files=2000),
|
| 368 |
+
GitHubRepo("vuejs", "vue", languages=["javascript", "typescript"], max_files=2000),
|
| 369 |
+
GitHubRepo("angular", "angular", languages=["typescript"], max_files=2000),
|
| 370 |
+
GitHubRepo("vercel", "next.js", languages=["javascript", "typescript"], max_files=2000),
|
| 371 |
+
GitHubRepo("microsoft", "TypeScript", languages=["typescript"], max_files=2000),
|
| 372 |
+
GitHubRepo("nodejs", "node", languages=["javascript", "cpp"], max_files=2000),
|
| 373 |
+
GitHubRepo("expressjs", "express", languages=["javascript"]),
|
| 374 |
+
GitHubRepo("lodash", "lodash", languages=["javascript"]),
|
| 375 |
+
GitHubRepo("axios", "axios", languages=["javascript"]),
|
| 376 |
+
GitHubRepo("chalk", "chalk", languages=["javascript"]),
|
| 377 |
+
|
| 378 |
+
# Go
|
| 379 |
+
GitHubRepo("golang", "go", languages=["go"], max_files=2000),
|
| 380 |
+
GitHubRepo("gin-gonic", "gin", languages=["go"]),
|
| 381 |
+
GitHubRepo("labstack", "echo", languages=["go"]),
|
| 382 |
+
GitHubRepo("spf13", "cobra", languages=["go"]),
|
| 383 |
+
GitHubRepo("kubernetes", "kubernetes", languages=["go"], max_files=2000),
|
| 384 |
+
GitHubRepo("prometheus", "prometheus", languages=["go"], max_files=2000),
|
| 385 |
+
GitHubRepo("grafana", "grafana", languages=["go"], max_files=2000),
|
| 386 |
+
GitHubRepo("etcd-io", "etcd", languages=["go"], max_files=2000),
|
| 387 |
+
GitHubRepo("hashicorp", "terraform", languages=["go"], max_files=2000),
|
| 388 |
+
GitHubRepo("hashicorp", "vault", languages=["go"], max_files=2000),
|
| 389 |
+
GitHubRepo("docker", "compose", languages=["go"]),
|
| 390 |
+
GitHubRepo("cli", "cli", languages=["go"]),
|
| 391 |
+
|
| 392 |
+
# Rust
|
| 393 |
+
GitHubRepo("rust-lang", "rust", languages=["rust"], max_files=2000),
|
| 394 |
+
GitHubRepo("rust-lang", "cargo", languages=["rust"], max_files=2000),
|
| 395 |
+
GitHubRepo("tokio-rs", "tokio", languages=["rust"], max_files=2000),
|
| 396 |
+
GitHubRepo("serde-rs", "serde", languages=["rust"]),
|
| 397 |
+
GitHubRepo("clap-rs", "clap", languages=["rust"]),
|
| 398 |
+
GitHubRepo("BurntSushi", "ripgrep", languages=["rust"]),
|
| 399 |
+
GitHubRepo("starship", "starship", languages=["rust"], max_files=2000),
|
| 400 |
+
|
| 401 |
+
# C/C++
|
| 402 |
+
GitHubRepo("redis", "redis", languages=["c"], max_files=2000),
|
| 403 |
+
GitHubRepo("sqlite", "sqlite", languages=["c"]),
|
| 404 |
+
GitHubRepo("curl", "curl", languages=["c"], max_files=2000),
|
| 405 |
+
GitHubRepo("nginx", "nginx", languages=["c"], max_files=2000),
|
| 406 |
+
GitHubRepo("openssl", "openssl", languages=["c"], max_files=2000),
|
| 407 |
+
|
| 408 |
+
# Tools/CLI
|
| 409 |
+
GitHubRepo("junegunn", "fzf", languages=["go"]),
|
| 410 |
+
GitHubRepo("BurntSushi", "ripgrep", languages=["rust"]),
|
| 411 |
+
GitHubRepo("sharkdp", "bat", languages=["rust"]),
|
| 412 |
+
GitHubRepo("sharkdp", "fd", languages=["rust"]),
|
| 413 |
+
GitHubRepo("dalance", "procs", languages=["rust"]),
|
| 414 |
+
|
| 415 |
+
# Documentation/Examples
|
| 416 |
+
GitHubRepo("realpython", "python-guide", languages=["python", "markdown"]),
|
| 417 |
+
GitHubRepo("ehmatthes", "pcc_2e", languages=["python"]),
|
| 418 |
+
GitHubRepo("thedaviddias", "Front-End-Checklist", languages=["markdown"]),
|
| 419 |
+
GitHubRepo("kamranahmedse", "developer-roadmap", languages=["markdown"]),
|
| 420 |
+
]
|
nexus/data/collectors/huggingface_collector.py
ADDED
|
@@ -0,0 +1,310 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
HuggingFace Collector - Thu thập datasets từ HuggingFace Hub
|
| 3 |
+
=============================================================
|
| 4 |
+
Pull datasets từ HuggingFace Hub cho training Nexus Coder.
|
| 5 |
+
|
| 6 |
+
Recommended datasets for code/text training:
|
| 7 |
+
- codeparrot/codeparrot-clean: Clean Python code
|
| 8 |
+
- GitHub CODE: Code from GitHub
|
| 9 |
+
- the-stack: Massive code dataset (3TB)
|
| 10 |
+
- oscar: Multilingual web text
|
| 11 |
+
- wikipedia: Wikipedia dumps
|
| 12 |
+
- openwebtext: Web text
|
| 13 |
+
- c4: Colossal Clean Crawled Corpus
|
| 14 |
+
- bookcorpus: Books
|
| 15 |
+
- arxiv: Scientific papers
|
| 16 |
+
- pubmed: Biomedical papers
|
| 17 |
+
"""
|
| 18 |
+
from __future__ import annotations
|
| 19 |
+
|
| 20 |
+
import os
|
| 21 |
+
import json
|
| 22 |
+
import logging
|
| 23 |
+
from typing import List, Dict, Optional, Iterator, Any
|
| 24 |
+
from dataclasses import dataclass, field
|
| 25 |
+
from pathlib import Path
|
| 26 |
+
|
| 27 |
+
logger = logging.getLogger(__name__)
|
| 28 |
+
|
| 29 |
+
|
| 30 |
+
@dataclass
|
| 31 |
+
class HFDataset:
|
| 32 |
+
"""Thông tin một HuggingFace dataset."""
|
| 33 |
+
name: str # e.g. "codeparrot/codeparrot-clean"
|
| 34 |
+
subset: Optional[str] = None
|
| 35 |
+
split: str = "train"
|
| 36 |
+
streaming: bool = True # Use streaming for large datasets
|
| 37 |
+
max_samples: int = 10000
|
| 38 |
+
field_mapping: Dict[str, str] = field(default_factory=lambda: {"text": "text"})
|
| 39 |
+
description: str = ""
|
| 40 |
+
language: Optional[str] = None # programming language for code datasets
|
| 41 |
+
size_gb: Optional[float] = None
|
| 42 |
+
|
| 43 |
+
|
| 44 |
+
# =============================================================================
|
| 45 |
+
# Curated list of high-quality datasets for Nexus Coder training
|
| 46 |
+
# =============================================================================
|
| 47 |
+
|
| 48 |
+
CURATED_DATASETS: List[HFDataset] = [
|
| 49 |
+
# === Code datasets ===
|
| 50 |
+
HFDataset(
|
| 51 |
+
name="codeparrot/codeparrot-clean",
|
| 52 |
+
max_samples=50000,
|
| 53 |
+
language="python",
|
| 54 |
+
description="Clean Python code from GitHub (preprocessed)",
|
| 55 |
+
size_gb=15,
|
| 56 |
+
),
|
| 57 |
+
HFDataset(
|
| 58 |
+
name="codeparrot/github-code",
|
| 59 |
+
max_samples=30000,
|
| 60 |
+
language="multiple",
|
| 61 |
+
description="Code from GitHub across multiple languages",
|
| 62 |
+
size_gb=115,
|
| 63 |
+
),
|
| 64 |
+
HFDataset(
|
| 65 |
+
name="bigcode/the-stack-dedup",
|
| 66 |
+
max_samples=20000,
|
| 67 |
+
language="multiple",
|
| 68 |
+
description="Deduplicated code from The Stack v2 (3TB)",
|
| 69 |
+
size_gb=3000,
|
| 70 |
+
),
|
| 71 |
+
HFDataset(
|
| 72 |
+
name="bigcode/the-stack-v2-train-full-ids",
|
| 73 |
+
max_samples=10000,
|
| 74 |
+
language="multiple",
|
| 75 |
+
description="The Stack v2 full training set",
|
| 76 |
+
size_gb=3000,
|
| 77 |
+
),
|
| 78 |
+
HFDataset(
|
| 79 |
+
name="nampdn-ai/tiny-codes",
|
| 80 |
+
max_samples=30000,
|
| 81 |
+
language="multiple",
|
| 82 |
+
description="Small high-quality code samples with instructions",
|
| 83 |
+
size_gb=2,
|
| 84 |
+
),
|
| 85 |
+
HFDataset(
|
| 86 |
+
name="HuggingFaceH4/CodeAlpaca_20K",
|
| 87 |
+
max_samples=20000,
|
| 88 |
+
language="python",
|
| 89 |
+
description="Code instruction dataset",
|
| 90 |
+
size_gb=0.1,
|
| 91 |
+
),
|
| 92 |
+
|
| 93 |
+
# === General text (Vietnamese + English) ===
|
| 94 |
+
HFDataset(
|
| 95 |
+
name="wikimedia/wikipedia",
|
| 96 |
+
subset="20231101.vi",
|
| 97 |
+
max_samples=20000,
|
| 98 |
+
description="Vietnamese Wikipedia",
|
| 99 |
+
size_gb=2,
|
| 100 |
+
),
|
| 101 |
+
HFDataset(
|
| 102 |
+
name="wikimedia/wikipedia",
|
| 103 |
+
subset="20231101.en",
|
| 104 |
+
max_samples=20000,
|
| 105 |
+
description="English Wikipedia",
|
| 106 |
+
size_gb=20,
|
| 107 |
+
),
|
| 108 |
+
HFDataset(
|
| 109 |
+
name="allenai/c4",
|
| 110 |
+
subset="multilingual",
|
| 111 |
+
split="train",
|
| 112 |
+
max_samples=10000,
|
| 113 |
+
description="Colossal Clean Crawled Corpus (multilingual)",
|
| 114 |
+
size_gb=25000,
|
| 115 |
+
),
|
| 116 |
+
HFDataset(
|
| 117 |
+
name="oscar-corpus/OSCAR-2301",
|
| 118 |
+
subset="vi",
|
| 119 |
+
max_samples=10000,
|
| 120 |
+
description="OSCAR Vietnamese web text",
|
| 121 |
+
size_gb=10,
|
| 122 |
+
),
|
| 123 |
+
|
| 124 |
+
# === Conversational / Instruction ===
|
| 125 |
+
HFDataset(
|
| 126 |
+
name="HuggingFaceH4/ultrachat_200k",
|
| 127 |
+
max_samples=20000,
|
| 128 |
+
description="High-quality multi-turn chat data",
|
| 129 |
+
size_gb=8,
|
| 130 |
+
),
|
| 131 |
+
HFDataset(
|
| 132 |
+
name="Open-Orca/OpenOrca",
|
| 133 |
+
max_samples=15000,
|
| 134 |
+
description="GPT-4 augmented FLAN instructions",
|
| 135 |
+
size_gb=50,
|
| 136 |
+
),
|
| 137 |
+
HFDataset(
|
| 138 |
+
name="teknium/OpenHermes-2.5",
|
| 139 |
+
max_samples=20000,
|
| 140 |
+
description="1M instruction samples",
|
| 141 |
+
size_gb=5,
|
| 142 |
+
),
|
| 143 |
+
HFDataset(
|
| 144 |
+
name="databricks/databricks-dolly-15k",
|
| 145 |
+
max_samples=15000,
|
| 146 |
+
description="Human-generated instruction data",
|
| 147 |
+
size_gb=0.2,
|
| 148 |
+
),
|
| 149 |
+
HFDataset(
|
| 150 |
+
name="allenai/RLVR-Chat",
|
| 151 |
+
max_samples=10000,
|
| 152 |
+
description="Reinforcement Learning from Verifiable Rewards chat data",
|
| 153 |
+
size_gb=2,
|
| 154 |
+
),
|
| 155 |
+
|
| 156 |
+
# === Math/Reasoning ===
|
| 157 |
+
HFDataset(
|
| 158 |
+
name="meta-math/MetaMathQA",
|
| 159 |
+
max_samples=20000,
|
| 160 |
+
description="Math Q&A with step-by-step solutions",
|
| 161 |
+
size_gb=1,
|
| 162 |
+
),
|
| 163 |
+
HFDataset(
|
| 164 |
+
name="gsm8k",
|
| 165 |
+
max_samples=8000,
|
| 166 |
+
description="Grade School Math 8K",
|
| 167 |
+
size_gb=0.01,
|
| 168 |
+
),
|
| 169 |
+
HFDataset(
|
| 170 |
+
name="lighteval/MATH",
|
| 171 |
+
max_samples=10000,
|
| 172 |
+
description="Competition math problems",
|
| 173 |
+
size_gb=0.05,
|
| 174 |
+
),
|
| 175 |
+
|
| 176 |
+
# === Scientific ===
|
| 177 |
+
HFDataset(
|
| 178 |
+
name="allenai/sciq",
|
| 179 |
+
max_samples=13000,
|
| 180 |
+
description="Science exam questions",
|
| 181 |
+
size_gb=0.05,
|
| 182 |
+
),
|
| 183 |
+
HFDataset(
|
| 184 |
+
name="allenai/openbookqa",
|
| 185 |
+
max_samples=5000,
|
| 186 |
+
description="Open-book science Q&A",
|
| 187 |
+
size_gb=0.02,
|
| 188 |
+
),
|
| 189 |
+
|
| 190 |
+
# === Vietnamese specific ===
|
| 191 |
+
HFDataset(
|
| 192 |
+
name="vietgpt/news_corpus",
|
| 193 |
+
max_samples=10000,
|
| 194 |
+
description="Vietnamese news corpus",
|
| 195 |
+
size_gb=2,
|
| 196 |
+
),
|
| 197 |
+
HFDataset(
|
| 198 |
+
name="PhoATC",
|
| 199 |
+
max_samples=5000,
|
| 200 |
+
description="Vietnamese text classification",
|
| 201 |
+
size_gb=0.1,
|
| 202 |
+
),
|
| 203 |
+
]
|
| 204 |
+
|
| 205 |
+
|
| 206 |
+
class HuggingFaceCollector:
|
| 207 |
+
"""Collect training data từ HuggingFace Hub.
|
| 208 |
+
|
| 209 |
+
Usage:
|
| 210 |
+
collector = HuggingFaceCollector(cache_dir="./data_cache/hf")
|
| 211 |
+
for sample in collector.collect(CURATED_DATASETS[:3]):
|
| 212 |
+
print(sample["text"][:100])
|
| 213 |
+
"""
|
| 214 |
+
|
| 215 |
+
def __init__(
|
| 216 |
+
self,
|
| 217 |
+
cache_dir: str = "./data_cache/hf",
|
| 218 |
+
token: Optional[str] = None,
|
| 219 |
+
):
|
| 220 |
+
self.cache_dir = cache_dir
|
| 221 |
+
self.token = token or os.environ.get("HF_TOKEN") or os.environ.get("HUGGING_FACE_HUB_TOKEN")
|
| 222 |
+
os.makedirs(cache_dir, exist_ok=True)
|
| 223 |
+
|
| 224 |
+
def collect(self, datasets: List[HFDataset]) -> Iterator[Dict[str, Any]]:
|
| 225 |
+
"""Collect samples từ list of HF datasets.
|
| 226 |
+
|
| 227 |
+
Yields:
|
| 228 |
+
Dict with keys: text, source, language, metadata
|
| 229 |
+
"""
|
| 230 |
+
try:
|
| 231 |
+
from datasets import load_dataset
|
| 232 |
+
except ImportError:
|
| 233 |
+
logger.error("datasets lib not installed. Run: pip install datasets")
|
| 234 |
+
return
|
| 235 |
+
|
| 236 |
+
for ds in datasets:
|
| 237 |
+
try:
|
| 238 |
+
yield from self._collect_dataset(ds, load_dataset)
|
| 239 |
+
except Exception as e:
|
| 240 |
+
logger.error(f"Failed to collect {ds.name}: {e}")
|
| 241 |
+
continue
|
| 242 |
+
|
| 243 |
+
def _collect_dataset(
|
| 244 |
+
self,
|
| 245 |
+
ds: HFDataset,
|
| 246 |
+
load_fn,
|
| 247 |
+
) -> Iterator[Dict[str, Any]]:
|
| 248 |
+
"""Collect từ một dataset."""
|
| 249 |
+
logger.info(f"Loading {ds.name} ({ds.subset or 'default'})...")
|
| 250 |
+
|
| 251 |
+
try:
|
| 252 |
+
if ds.streaming:
|
| 253 |
+
dataset = load_fn(
|
| 254 |
+
ds.name,
|
| 255 |
+
name=ds.subset,
|
| 256 |
+
split=ds.split,
|
| 257 |
+
streaming=True,
|
| 258 |
+
token=self.token,
|
| 259 |
+
)
|
| 260 |
+
else:
|
| 261 |
+
dataset = load_fn(
|
| 262 |
+
ds.name,
|
| 263 |
+
name=ds.subset,
|
| 264 |
+
split=ds.split,
|
| 265 |
+
token=self.token,
|
| 266 |
+
cache_dir=self.cache_dir,
|
| 267 |
+
)
|
| 268 |
+
except Exception as e:
|
| 269 |
+
logger.error(f"Failed to load {ds.name}: {e}")
|
| 270 |
+
return
|
| 271 |
+
|
| 272 |
+
count = 0
|
| 273 |
+
text_field = ds.field_mapping.get("text", "text")
|
| 274 |
+
|
| 275 |
+
for item in dataset:
|
| 276 |
+
if count >= ds.max_samples:
|
| 277 |
+
break
|
| 278 |
+
|
| 279 |
+
# Extract text using field mapping
|
| 280 |
+
text = item.get(text_field) or item.get("text") or item.get("content") or ""
|
| 281 |
+
|
| 282 |
+
if not text or not isinstance(text, str):
|
| 283 |
+
# Try concatenating fields
|
| 284 |
+
text = " ".join(str(v) for v in item.values() if isinstance(v, str))
|
| 285 |
+
|
| 286 |
+
if not text or len(text) < 50:
|
| 287 |
+
continue
|
| 288 |
+
|
| 289 |
+
yield {
|
| 290 |
+
"text": text,
|
| 291 |
+
"source": ds.name,
|
| 292 |
+
"language": ds.language or "text",
|
| 293 |
+
"metadata": {
|
| 294 |
+
"dataset": ds.name,
|
| 295 |
+
"subset": ds.subset,
|
| 296 |
+
"split": ds.split,
|
| 297 |
+
"original_size": len(text),
|
| 298 |
+
},
|
| 299 |
+
}
|
| 300 |
+
count += 1
|
| 301 |
+
|
| 302 |
+
logger.info(f"Collected {count} samples from {ds.name}")
|
| 303 |
+
|
| 304 |
+
def list_available(self) -> List[HFDataset]:
|
| 305 |
+
"""Return curated list of datasets."""
|
| 306 |
+
return CURATED_DATASETS
|
| 307 |
+
|
| 308 |
+
def estimate_total_size(self, datasets: List[HFDataset]) -> float:
|
| 309 |
+
"""Estimate total size in GB."""
|
| 310 |
+
return sum(ds.size_gb or 0 for ds in datasets)
|
nexus/data/collectors/python_alpaca_collector.py
ADDED
|
@@ -0,0 +1,117 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
Python-Alpaca Collector for Nexus Coder v0.3
|
| 3 |
+
=============================================
|
| 4 |
+
Aggregates multiple high-quality Python instruction-tuning datasets.
|
| 5 |
+
|
| 6 |
+
Sources (all on HuggingFace):
|
| 7 |
+
- sahil2801/codealpaca ~20K samples
|
| 8 |
+
- HuggingFaceH4/CodeAlpaca_20K ~20K
|
| 9 |
+
- nickroany/Evol-Instruct-Code ~15K
|
| 10 |
+
- TheBloke/CodeAlpaca-13B ~5K
|
| 11 |
+
- codeparrot/codeparrot-clean ~50K (filterable)
|
| 12 |
+
- nampdn-ai/tiny-codes ~50K (filterable)
|
| 13 |
+
|
| 14 |
+
Output: unified JSONL with Nexus format {system, user, assistant}.
|
| 15 |
+
Converts Alpaca-style {instruction, input, output} → unified via
|
| 16 |
+
nexus.integrations.llamafactory.alpaca_to_nexus.
|
| 17 |
+
|
| 18 |
+
Author: Hieu Louis (2026)
|
| 19 |
+
"""
|
| 20 |
+
from __future__ import annotations
|
| 21 |
+
|
| 22 |
+
import json
|
| 23 |
+
import os
|
| 24 |
+
from typing import Dict, Iterator, List, Optional
|
| 25 |
+
|
| 26 |
+
# We import the converter for type hints only — actual import at runtime
|
| 27 |
+
# to keep the module importable when llamafactory deps are missing.
|
| 28 |
+
try:
|
| 29 |
+
from ...integrations.llamafactory import convert_to_nexus
|
| 30 |
+
_HAS_CONVERTER = True
|
| 31 |
+
except Exception:
|
| 32 |
+
_HAS_CONVERTER = False
|
| 33 |
+
|
| 34 |
+
|
| 35 |
+
DEFAULT_SOURCES = [
|
| 36 |
+
{"name": "sahil2801/codealpaca", "max_samples": 20000},
|
| 37 |
+
{"name": "HuggingFaceH4/CodeAlpaca_20K", "max_samples": 20000},
|
| 38 |
+
{"name": "nickroany/Evol-Instruct-Code", "max_samples": 15000},
|
| 39 |
+
{"name": "TheBloke/CodeAlpaca-13B", "max_samples": 5000},
|
| 40 |
+
{"name": "codeparrot/codeparrot-clean", "max_samples": 50000, "is_completion": True},
|
| 41 |
+
{"name": "nampdn-ai/tiny-codes", "max_samples": 50000},
|
| 42 |
+
]
|
| 43 |
+
|
| 44 |
+
|
| 45 |
+
class PythonAlpacaCollector:
|
| 46 |
+
"""Aggregate Python instruction datasets."""
|
| 47 |
+
|
| 48 |
+
def __init__(
|
| 49 |
+
self,
|
| 50 |
+
cache_dir: str = "./data_cache/python_alpaca",
|
| 51 |
+
sources: Optional[List[Dict]] = None,
|
| 52 |
+
):
|
| 53 |
+
self.cache_dir = cache_dir
|
| 54 |
+
self.sources = sources or DEFAULT_SOURCES
|
| 55 |
+
os.makedirs(cache_dir, exist_ok=True)
|
| 56 |
+
|
| 57 |
+
def _iter_source(self, source: Dict) -> Iterator[Dict]:
|
| 58 |
+
name = source["name"]
|
| 59 |
+
max_samples = source.get("max_samples", 10000)
|
| 60 |
+
is_completion = source.get("is_completion", False)
|
| 61 |
+
try:
|
| 62 |
+
from datasets import load_dataset
|
| 63 |
+
except ImportError:
|
| 64 |
+
return
|
| 65 |
+
try:
|
| 66 |
+
ds = load_dataset(name, split="train", streaming=True)
|
| 67 |
+
except Exception:
|
| 68 |
+
return
|
| 69 |
+
count = 0
|
| 70 |
+
for example in ds:
|
| 71 |
+
if count >= max_samples:
|
| 72 |
+
break
|
| 73 |
+
# Normalize to Nexus format
|
| 74 |
+
try:
|
| 75 |
+
if _HAS_CONVERTER:
|
| 76 |
+
turns = convert_to_nexus(example)
|
| 77 |
+
else:
|
| 78 |
+
# Inline fallback for Alpaca format
|
| 79 |
+
turns = [{
|
| 80 |
+
"system": example.get("system_prompt", ""),
|
| 81 |
+
"user": example.get("instruction", ""),
|
| 82 |
+
"assistant": example.get("output", ""),
|
| 83 |
+
}]
|
| 84 |
+
for turn in turns:
|
| 85 |
+
if not turn.get("user") or not turn.get("assistant"):
|
| 86 |
+
continue
|
| 87 |
+
yield {
|
| 88 |
+
"source": name,
|
| 89 |
+
"system": turn.get("system", ""),
|
| 90 |
+
"user": turn["user"],
|
| 91 |
+
"assistant": turn["assistant"],
|
| 92 |
+
}
|
| 93 |
+
count += 1
|
| 94 |
+
if count >= max_samples:
|
| 95 |
+
break
|
| 96 |
+
except Exception:
|
| 97 |
+
continue
|
| 98 |
+
|
| 99 |
+
def __iter__(self) -> Iterator[Dict]:
|
| 100 |
+
for source in self.sources:
|
| 101 |
+
yield from self._iter_source(source)
|
| 102 |
+
|
| 103 |
+
def collect(self, output_dir: Optional[str] = None) -> str:
|
| 104 |
+
"""Collect and write JSONL. Returns output path."""
|
| 105 |
+
output_dir = output_dir or self.cache_dir
|
| 106 |
+
os.makedirs(output_dir, exist_ok=True)
|
| 107 |
+
output_path = os.path.join(output_dir, "python_alpaca.jsonl")
|
| 108 |
+
total = 0
|
| 109 |
+
with open(output_path, "w", encoding="utf-8") as f:
|
| 110 |
+
for sample in self:
|
| 111 |
+
f.write(json.dumps(sample, ensure_ascii=False) + "\n")
|
| 112 |
+
total += 1
|
| 113 |
+
print(f"[PythonAlpacaCollector] Collected {total} samples → {output_path}")
|
| 114 |
+
return output_path
|
| 115 |
+
|
| 116 |
+
|
| 117 |
+
__all__ = ["PythonAlpacaCollector", "DEFAULT_SOURCES"]
|
nexus/data/collectors/stackoverflow_collector.py
ADDED
|
@@ -0,0 +1,250 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
StackOverflow Collector - Thu thập Q&A từ StackOverflow
|
| 3 |
+
=========================================================
|
| 4 |
+
"""
|
| 5 |
+
from __future__ import annotations
|
| 6 |
+
|
| 7 |
+
import logging
|
| 8 |
+
import urllib.request
|
| 9 |
+
import urllib.parse
|
| 10 |
+
import json
|
| 11 |
+
import time
|
| 12 |
+
from typing import List, Dict, Optional, Iterator, Any
|
| 13 |
+
from dataclasses import dataclass
|
| 14 |
+
|
| 15 |
+
logger = logging.getLogger(__name__)
|
| 16 |
+
|
| 17 |
+
|
| 18 |
+
@dataclass
|
| 19 |
+
class SOQuestion:
|
| 20 |
+
"""Một StackOverflow question."""
|
| 21 |
+
question_id: int
|
| 22 |
+
title: str
|
| 23 |
+
body: str
|
| 24 |
+
tags: List[str]
|
| 25 |
+
score: int
|
| 26 |
+
answer_count: int
|
| 27 |
+
accepted_answer_id: Optional[int] = None
|
| 28 |
+
answers: List[Dict] = None
|
| 29 |
+
|
| 30 |
+
|
| 31 |
+
# v0.4 fix: expose at module level (was inside the class, broke `from ... import CURATED_TAGS`)
|
| 32 |
+
CURATED_TAGS = [
|
| 33 |
+
"python", "javascript", "java", "c#", "php", "android",
|
| 34 |
+
"html", "jquery", "c++", "css", "ios", "mysql",
|
| 35 |
+
"sql", "node.js", "reactjs", "ruby-on-rails", "vue.js",
|
| 36 |
+
"typescript", "docker", "git", "go", "rust",
|
| 37 |
+
"machine-learning", "deep-learning", "pytorch", "tensorflow",
|
| 38 |
+
"pandas", "numpy", "regex", "algorithm", "data-structures",
|
| 39 |
+
"unit-testing", "debugging", "performance", "security",
|
| 40 |
+
]
|
| 41 |
+
|
| 42 |
+
|
| 43 |
+
class StackOverflowCollector:
|
| 44 |
+
"""Collect Q&A từ StackOverflow API.
|
| 45 |
+
|
| 46 |
+
StackOverflow API: 10000 requests/day without key, 50000 with key.
|
| 47 |
+
Rate limit: 30 requests/second.
|
| 48 |
+
"""
|
| 49 |
+
|
| 50 |
+
BASE_URL = "https://api.stackexchange.com/2.3"
|
| 51 |
+
|
| 52 |
+
# Backward-compat alias (deprecation: prefer module-level CURATED_TAGS)
|
| 53 |
+
CURATED_TAGS = CURATED_TAGS
|
| 54 |
+
|
| 55 |
+
def __init__(
|
| 56 |
+
self,
|
| 57 |
+
key: Optional[str] = None,
|
| 58 |
+
access_token: Optional[str] = None,
|
| 59 |
+
page_size: int = 100,
|
| 60 |
+
):
|
| 61 |
+
self.key = key
|
| 62 |
+
self.access_token = access_token
|
| 63 |
+
self.page_size = min(page_size, 100)
|
| 64 |
+
|
| 65 |
+
def search(
|
| 66 |
+
self,
|
| 67 |
+
tag: str,
|
| 68 |
+
max_results: int = 500,
|
| 69 |
+
min_score: int = 5,
|
| 70 |
+
sort: str = "votes",
|
| 71 |
+
) -> List[SOQuestion]:
|
| 72 |
+
"""Search questions by tag.
|
| 73 |
+
|
| 74 |
+
Args:
|
| 75 |
+
tag: Tag to filter (e.g. "python")
|
| 76 |
+
max_results: Max questions to return
|
| 77 |
+
min_score: Minimum question score
|
| 78 |
+
sort: "votes", "creation", "activity"
|
| 79 |
+
"""
|
| 80 |
+
questions = []
|
| 81 |
+
page = 1
|
| 82 |
+
|
| 83 |
+
while len(questions) < max_results and page <= 50: # API limit: 50 pages
|
| 84 |
+
params = {
|
| 85 |
+
"order": "desc",
|
| 86 |
+
"sort": sort,
|
| 87 |
+
"tagged": tag,
|
| 88 |
+
"site": "stackoverflow",
|
| 89 |
+
"pagesize": str(self.page_size),
|
| 90 |
+
"page": str(page),
|
| 91 |
+
"filter": "withbody", # Include body
|
| 92 |
+
"min": str(min_score),
|
| 93 |
+
}
|
| 94 |
+
if self.key:
|
| 95 |
+
params["key"] = self.key
|
| 96 |
+
if self.access_token:
|
| 97 |
+
params["access_token"] = self.access_token
|
| 98 |
+
|
| 99 |
+
url = f"{self.BASE_URL}/questions?{urllib.parse.urlencode(params)}"
|
| 100 |
+
|
| 101 |
+
try:
|
| 102 |
+
req = urllib.request.Request(url, headers={
|
| 103 |
+
"Accept-Encoding": "gzip",
|
| 104 |
+
"User-Agent": "NexusCoder-Collector/0.2",
|
| 105 |
+
})
|
| 106 |
+
with urllib.request.urlopen(req, timeout=30) as response:
|
| 107 |
+
# Handle gzip
|
| 108 |
+
if response.headers.get("Content-Encoding") == "gzip":
|
| 109 |
+
import gzip
|
| 110 |
+
data = json.loads(gzip.decompress(response.read()).decode())
|
| 111 |
+
else:
|
| 112 |
+
data = json.loads(response.read().decode())
|
| 113 |
+
|
| 114 |
+
items = data.get("items", [])
|
| 115 |
+
if not items:
|
| 116 |
+
break
|
| 117 |
+
|
| 118 |
+
for item in items:
|
| 119 |
+
questions.append(SOQuestion(
|
| 120 |
+
question_id=item["question_id"],
|
| 121 |
+
title=item["title"],
|
| 122 |
+
body=item.get("body", ""),
|
| 123 |
+
tags=item.get("tags", []),
|
| 124 |
+
score=item.get("score", 0),
|
| 125 |
+
answer_count=item.get("answer_count", 0),
|
| 126 |
+
accepted_answer_id=item.get("accepted_answer_id"),
|
| 127 |
+
))
|
| 128 |
+
|
| 129 |
+
# Check if more pages
|
| 130 |
+
if not data.get("has_more", False):
|
| 131 |
+
break
|
| 132 |
+
|
| 133 |
+
# Backoff if needed
|
| 134 |
+
if data.get("backoff"):
|
| 135 |
+
time.sleep(data["backoff"])
|
| 136 |
+
|
| 137 |
+
page += 1
|
| 138 |
+
time.sleep(0.5) # Polite delay
|
| 139 |
+
|
| 140 |
+
except Exception as e:
|
| 141 |
+
logger.error(f"SO search failed: {e}")
|
| 142 |
+
break
|
| 143 |
+
|
| 144 |
+
return questions[:max_results]
|
| 145 |
+
|
| 146 |
+
def get_answers(self, question_ids: List[int]) -> Dict[int, List[Dict]]:
|
| 147 |
+
"""Get answers for multiple questions."""
|
| 148 |
+
if not question_ids:
|
| 149 |
+
return {}
|
| 150 |
+
|
| 151 |
+
ids_str = ";".join(str(qid) for qid in question_ids[:100]) # Max 100 ids
|
| 152 |
+
params = {
|
| 153 |
+
"order": "desc",
|
| 154 |
+
"sort": "votes",
|
| 155 |
+
"site": "stackoverflow",
|
| 156 |
+
"filter": "withbody",
|
| 157 |
+
}
|
| 158 |
+
if self.key:
|
| 159 |
+
params["key"] = self.key
|
| 160 |
+
|
| 161 |
+
url = f"{self.BASE_URL}/questions/{ids_str}/answers?{urllib.parse.urlencode(params)}"
|
| 162 |
+
|
| 163 |
+
try:
|
| 164 |
+
req = urllib.request.Request(url, headers={
|
| 165 |
+
"Accept-Encoding": "gzip",
|
| 166 |
+
"User-Agent": "NexusCoder-Collector/0.2",
|
| 167 |
+
})
|
| 168 |
+
with urllib.request.urlopen(req, timeout=30) as response:
|
| 169 |
+
if response.headers.get("Content-Encoding") == "gzip":
|
| 170 |
+
import gzip
|
| 171 |
+
data = json.loads(gzip.decompress(response.read()).decode())
|
| 172 |
+
else:
|
| 173 |
+
data = json.loads(response.read().decode())
|
| 174 |
+
|
| 175 |
+
answers_by_q = {}
|
| 176 |
+
for ans in data.get("items", []):
|
| 177 |
+
qid = ans["question_id"]
|
| 178 |
+
if qid not in answers_by_q:
|
| 179 |
+
answers_by_q[qid] = []
|
| 180 |
+
answers_by_q[qid].append({
|
| 181 |
+
"answer_id": ans["answer_id"],
|
| 182 |
+
"body": ans.get("body", ""),
|
| 183 |
+
"score": ans.get("score", 0),
|
| 184 |
+
"is_accepted": ans.get("is_accepted", False),
|
| 185 |
+
})
|
| 186 |
+
|
| 187 |
+
return answers_by_q
|
| 188 |
+
except Exception as e:
|
| 189 |
+
logger.error(f"SO get_answers failed: {e}")
|
| 190 |
+
return {}
|
| 191 |
+
|
| 192 |
+
def collect(
|
| 193 |
+
self,
|
| 194 |
+
tags: Optional[List[str]] = None,
|
| 195 |
+
max_per_tag: int = 100,
|
| 196 |
+
include_answers: bool = True,
|
| 197 |
+
) -> Iterator[Dict[str, Any]]:
|
| 198 |
+
"""Collect Q&A pairs as training samples.
|
| 199 |
+
|
| 200 |
+
Yields:
|
| 201 |
+
Dict with keys: text (formatted Q&A), source, language, metadata
|
| 202 |
+
"""
|
| 203 |
+
tags = tags or self.CURATED_TAGS[:10]
|
| 204 |
+
|
| 205 |
+
for tag in tags:
|
| 206 |
+
logger.info(f"Collecting SO tag: {tag}")
|
| 207 |
+
questions = self.search(tag, max_results=max_per_tag)
|
| 208 |
+
|
| 209 |
+
if include_answers and questions:
|
| 210 |
+
qids = [q.question_id for q in questions if q.accepted_answer_id]
|
| 211 |
+
answers_by_q = self.get_answers(qids)
|
| 212 |
+
else:
|
| 213 |
+
answers_by_q = {}
|
| 214 |
+
|
| 215 |
+
for q in questions:
|
| 216 |
+
# Format as Q&A pair
|
| 217 |
+
answer_text = ""
|
| 218 |
+
if q.question_id in answers_by_q:
|
| 219 |
+
accepted = [a for a in answers_by_q[q.question_id] if a["is_accepted"]]
|
| 220 |
+
if accepted:
|
| 221 |
+
answer_text = accepted[0]["body"]
|
| 222 |
+
elif answers_by_q[q.question_id]:
|
| 223 |
+
answer_text = answers_by_q[q.question_id][0]["body"]
|
| 224 |
+
|
| 225 |
+
if not answer_text:
|
| 226 |
+
continue
|
| 227 |
+
|
| 228 |
+
# Strip HTML tags (simple)
|
| 229 |
+
import re
|
| 230 |
+
q_body_clean = re.sub(r"<[^>]+>", "", q.body)
|
| 231 |
+
a_clean = re.sub(r"<[^>]+>", "", answer_text)
|
| 232 |
+
|
| 233 |
+
text = (
|
| 234 |
+
f"Question: {q.title}\n\n"
|
| 235 |
+
f"Tags: {', '.join(q.tags)}\n\n"
|
| 236 |
+
f"{q_body_clean}\n\n"
|
| 237 |
+
f"Answer:\n{a_clean}"
|
| 238 |
+
)
|
| 239 |
+
|
| 240 |
+
yield {
|
| 241 |
+
"text": text,
|
| 242 |
+
"source": "stackoverflow",
|
| 243 |
+
"language": "en",
|
| 244 |
+
"metadata": {
|
| 245 |
+
"question_id": q.question_id,
|
| 246 |
+
"tags": q.tags,
|
| 247 |
+
"score": q.score,
|
| 248 |
+
"title": q.title,
|
| 249 |
+
},
|
| 250 |
+
}
|
nexus/data/collectors/starcoder2_collector.py
ADDED
|
@@ -0,0 +1,186 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
StarCoder2-data Collector for Nexus Coder v0.3
|
| 3 |
+
===============================================
|
| 4 |
+
Pulls from BigCode's StarCoder2 training data (github-code, commits, jupyter).
|
| 5 |
+
|
| 6 |
+
Components:
|
| 7 |
+
- github_code: raw code files from GitHub (subset of The-Stack v2)
|
| 8 |
+
- github_commits: commit diffs — great for code-editing / instruction tasks
|
| 9 |
+
- github_jupyter: notebook cells with markdown + code interleaved
|
| 10 |
+
|
| 11 |
+
Each component has different schema; this collector unifies them into the
|
| 12 |
+
Nexus format: {source, lang, content, metadata}.
|
| 13 |
+
|
| 14 |
+
Reference:
|
| 15 |
+
BigCode. "StarCoder 2 and The Stack v2: Building the Next Generation of
|
| 16 |
+
Transparent Code Models."
|
| 17 |
+
https://huggingface.co/datasets/bigcode/starcoder2data
|
| 18 |
+
|
| 19 |
+
Author: Hieu Louis (2026)
|
| 20 |
+
"""
|
| 21 |
+
from __future__ import annotations
|
| 22 |
+
|
| 23 |
+
import json
|
| 24 |
+
import os
|
| 25 |
+
from typing import Dict, Iterator, List, Optional
|
| 26 |
+
|
| 27 |
+
|
| 28 |
+
COMPONENT_DATASETS = {
|
| 29 |
+
"github_code": "bigcode/starcoder2data",
|
| 30 |
+
"github_commits": "bigcode/starcoder2data",
|
| 31 |
+
"github_jupyter": "bigcode/starcoder2data",
|
| 32 |
+
}
|
| 33 |
+
|
| 34 |
+
SUPPORTED_LANGS = [
|
| 35 |
+
"python", "javascript", "typescript", "java",
|
| 36 |
+
"go", "rust", "c", "cpp",
|
| 37 |
+
]
|
| 38 |
+
|
| 39 |
+
|
| 40 |
+
class StarCoder2Collector:
|
| 41 |
+
"""Collect from StarCoder2 training data."""
|
| 42 |
+
|
| 43 |
+
def __init__(
|
| 44 |
+
self,
|
| 45 |
+
cache_dir: str = "./data_cache/starcoder2",
|
| 46 |
+
components: Optional[List[str]] = None,
|
| 47 |
+
max_samples_per_component: int = 20000,
|
| 48 |
+
languages: Optional[List[str]] = None,
|
| 49 |
+
streaming: bool = True,
|
| 50 |
+
):
|
| 51 |
+
self.cache_dir = cache_dir
|
| 52 |
+
self.components = components or list(COMPONENT_DATASETS.keys())
|
| 53 |
+
self.max_samples_per_component = max_samples_per_component
|
| 54 |
+
self.languages = languages or SUPPORTED_LANGS
|
| 55 |
+
self.streaming = streaming
|
| 56 |
+
os.makedirs(cache_dir, exist_ok=True)
|
| 57 |
+
|
| 58 |
+
def _iter_github_code(self) -> Iterator[Dict]:
|
| 59 |
+
"""Iterate github_code component."""
|
| 60 |
+
try:
|
| 61 |
+
from datasets import load_dataset
|
| 62 |
+
except ImportError:
|
| 63 |
+
return
|
| 64 |
+
for lang in self.languages:
|
| 65 |
+
count = 0
|
| 66 |
+
try:
|
| 67 |
+
ds = load_dataset(
|
| 68 |
+
"bigcode/starcoder2data",
|
| 69 |
+
split="train",
|
| 70 |
+
streaming=self.streaming,
|
| 71 |
+
data_dir=f"data/{lang}",
|
| 72 |
+
)
|
| 73 |
+
except Exception:
|
| 74 |
+
continue
|
| 75 |
+
for example in ds:
|
| 76 |
+
if count >= self.max_samples_per_component // len(self.languages):
|
| 77 |
+
break
|
| 78 |
+
content = example.get("content", "")
|
| 79 |
+
if not content or len(content) < 50:
|
| 80 |
+
continue
|
| 81 |
+
yield {
|
| 82 |
+
"source": "starcoder2_github_code",
|
| 83 |
+
"lang": lang,
|
| 84 |
+
"content": content,
|
| 85 |
+
"metadata": {
|
| 86 |
+
"repo": example.get("repository", ""),
|
| 87 |
+
"path": example.get("path", ""),
|
| 88 |
+
"size": example.get("size", 0),
|
| 89 |
+
"license": example.get("license", ""),
|
| 90 |
+
},
|
| 91 |
+
}
|
| 92 |
+
count += 1
|
| 93 |
+
|
| 94 |
+
def _iter_github_commits(self) -> Iterator[Dict]:
|
| 95 |
+
"""Iterate github_commits component (commit diffs)."""
|
| 96 |
+
try:
|
| 97 |
+
from datasets import load_dataset
|
| 98 |
+
except ImportError:
|
| 99 |
+
return
|
| 100 |
+
count = 0
|
| 101 |
+
try:
|
| 102 |
+
ds = load_dataset(
|
| 103 |
+
"bigcode/starcoder2data",
|
| 104 |
+
split="train",
|
| 105 |
+
streaming=self.streaming,
|
| 106 |
+
name="commits",
|
| 107 |
+
)
|
| 108 |
+
except Exception:
|
| 109 |
+
return
|
| 110 |
+
for example in ds:
|
| 111 |
+
if count >= self.max_samples_per_component:
|
| 112 |
+
break
|
| 113 |
+
diff = example.get("diff", "") or example.get("content", "")
|
| 114 |
+
if not diff or len(diff) < 50:
|
| 115 |
+
continue
|
| 116 |
+
yield {
|
| 117 |
+
"source": "starcoder2_commits",
|
| 118 |
+
"lang": example.get("language", "unknown"),
|
| 119 |
+
"content": diff,
|
| 120 |
+
"metadata": {
|
| 121 |
+
"commit": example.get("commit", ""),
|
| 122 |
+
"repo": example.get("repository", ""),
|
| 123 |
+
"author": example.get("author", ""),
|
| 124 |
+
},
|
| 125 |
+
}
|
| 126 |
+
count += 1
|
| 127 |
+
|
| 128 |
+
def _iter_github_jupyter(self) -> Iterator[Dict]:
|
| 129 |
+
"""Iterate github_jupyter component (notebook cells)."""
|
| 130 |
+
try:
|
| 131 |
+
from datasets import load_dataset
|
| 132 |
+
except ImportError:
|
| 133 |
+
return
|
| 134 |
+
count = 0
|
| 135 |
+
try:
|
| 136 |
+
ds = load_dataset(
|
| 137 |
+
"bigcode/starcoder2data",
|
| 138 |
+
split="train",
|
| 139 |
+
streaming=self.streaming,
|
| 140 |
+
name="jupyter",
|
| 141 |
+
)
|
| 142 |
+
except Exception:
|
| 143 |
+
return
|
| 144 |
+
for example in ds:
|
| 145 |
+
if count >= self.max_samples_per_component:
|
| 146 |
+
break
|
| 147 |
+
content = example.get("content", "")
|
| 148 |
+
if not content or len(content) < 50:
|
| 149 |
+
continue
|
| 150 |
+
yield {
|
| 151 |
+
"source": "starcoder2_jupyter",
|
| 152 |
+
"lang": "python",
|
| 153 |
+
"content": content,
|
| 154 |
+
"metadata": {
|
| 155 |
+
"repo": example.get("repository", ""),
|
| 156 |
+
"notebook_path": example.get("path", ""),
|
| 157 |
+
"cell_type": example.get("cell_type", ""),
|
| 158 |
+
},
|
| 159 |
+
}
|
| 160 |
+
count += 1
|
| 161 |
+
|
| 162 |
+
def __iter__(self) -> Iterator[Dict]:
|
| 163 |
+
"""Stream samples from all enabled components."""
|
| 164 |
+
for component in self.components:
|
| 165 |
+
if component == "github_code":
|
| 166 |
+
yield from self._iter_github_code()
|
| 167 |
+
elif component == "github_commits":
|
| 168 |
+
yield from self._iter_github_commits()
|
| 169 |
+
elif component == "github_jupyter":
|
| 170 |
+
yield from self._iter_github_jupyter()
|
| 171 |
+
|
| 172 |
+
def collect(self, output_dir: Optional[str] = None) -> str:
|
| 173 |
+
"""Collect all samples and write to JSONL. Returns the output file path."""
|
| 174 |
+
output_dir = output_dir or self.cache_dir
|
| 175 |
+
os.makedirs(output_dir, exist_ok=True)
|
| 176 |
+
output_path = os.path.join(output_dir, "starcoder2.jsonl")
|
| 177 |
+
total = 0
|
| 178 |
+
with open(output_path, "w", encoding="utf-8") as f:
|
| 179 |
+
for sample in self:
|
| 180 |
+
f.write(json.dumps(sample, ensure_ascii=False) + "\n")
|
| 181 |
+
total += 1
|
| 182 |
+
print(f"[StarCoder2Collector] Collected {total} samples → {output_path}")
|
| 183 |
+
return output_path
|
| 184 |
+
|
| 185 |
+
|
| 186 |
+
__all__ = ["StarCoder2Collector", "COMPONENT_DATASETS", "SUPPORTED_LANGS"]
|
nexus/data/collectors/the_stack_collector.py
ADDED
|
@@ -0,0 +1,131 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
The-Stack v2 Collector for Nexus Coder v0.3
|
| 3 |
+
============================================
|
| 4 |
+
Pulls code samples from BigCode's The-Stack v2 dataset on HuggingFace.
|
| 5 |
+
|
| 6 |
+
The-Stack v2 is a massive deduplicated code corpus covering ~600 programming
|
| 7 |
+
languages, collected from GitHub repos with permissive licenses.
|
| 8 |
+
|
| 9 |
+
This collector:
|
| 10 |
+
- Streams samples lazily via `datasets` library (lazy import)
|
| 11 |
+
- Filters by language (Python, JS, TS, Go, Rust, etc.)
|
| 12 |
+
- Applies license filter (only MIT/Apache/BSD/MPL)
|
| 13 |
+
- Writes to JSONL with metadata {lang, license, repo, path, content}
|
| 14 |
+
|
| 15 |
+
Reference:
|
| 16 |
+
BigCode. "The Stack v2: A Comprehensive Multilingual Code Corpus."
|
| 17 |
+
https://huggingface.co/datasets/bigcode/the-stack-v2-train-full-ids
|
| 18 |
+
|
| 19 |
+
Author: Hieu Louis (2026)
|
| 20 |
+
"""
|
| 21 |
+
from __future__ import annotations
|
| 22 |
+
|
| 23 |
+
import json
|
| 24 |
+
import os
|
| 25 |
+
from typing import Dict, Iterator, List, Optional
|
| 26 |
+
|
| 27 |
+
|
| 28 |
+
# Curated language list (subset of v2's ~600 languages)
|
| 29 |
+
SUPPORTED_LANGUAGES = [
|
| 30 |
+
"python", "javascript", "typescript", "java", "go", "rust",
|
| 31 |
+
"c", "cpp", "csharp", "ruby", "php", "swift", "kotlin",
|
| 32 |
+
"scala", "shell", "sql", "html", "css",
|
| 33 |
+
]
|
| 34 |
+
|
| 35 |
+
# Permissive licenses (allowlist)
|
| 36 |
+
PERMISSIVE_LICENSES = {
|
| 37 |
+
"mit", "apache-2.0", "bsd-3-clause", "bsd-2-clause",
|
| 38 |
+
"mpl-2.0", "unlicense", "isc", "0bsd",
|
| 39 |
+
}
|
| 40 |
+
|
| 41 |
+
|
| 42 |
+
class TheStackCollector:
|
| 43 |
+
"""Collect code samples from The-Stack v2."""
|
| 44 |
+
|
| 45 |
+
DATASET_NAME = "bigcode/the-stack-v2-train-full-ids"
|
| 46 |
+
|
| 47 |
+
def __init__(
|
| 48 |
+
self,
|
| 49 |
+
cache_dir: str = "./data_cache/the_stack",
|
| 50 |
+
languages: Optional[List[str]] = None,
|
| 51 |
+
max_samples_per_language: int = 5000,
|
| 52 |
+
min_stars: int = 0,
|
| 53 |
+
license_filter: Optional[List[str]] = None,
|
| 54 |
+
streaming: bool = True,
|
| 55 |
+
):
|
| 56 |
+
self.cache_dir = cache_dir
|
| 57 |
+
self.languages = languages or SUPPORTED_LANGUAGES
|
| 58 |
+
self.max_samples_per_language = max_samples_per_language
|
| 59 |
+
self.min_stars = min_stars
|
| 60 |
+
self.license_filter = set(license_filter) if license_filter else PERMISSIVE_LICENSES
|
| 61 |
+
self.streaming = streaming
|
| 62 |
+
os.makedirs(cache_dir, exist_ok=True)
|
| 63 |
+
|
| 64 |
+
def __iter__(self) -> Iterator[Dict]:
|
| 65 |
+
"""Stream samples lazily from The-Stack v2.
|
| 66 |
+
Yields dicts: {lang, license, repo, path, size, content}.
|
| 67 |
+
"""
|
| 68 |
+
try:
|
| 69 |
+
from datasets import load_dataset # lazy import
|
| 70 |
+
except ImportError as e:
|
| 71 |
+
raise ImportError(
|
| 72 |
+
"The `datasets` package is required. Install with: pip install datasets"
|
| 73 |
+
) from e
|
| 74 |
+
|
| 75 |
+
for lang in self.languages:
|
| 76 |
+
count = 0
|
| 77 |
+
try:
|
| 78 |
+
ds = load_dataset(
|
| 79 |
+
self.DATASET_NAME,
|
| 80 |
+
split="train",
|
| 81 |
+
streaming=self.streaming,
|
| 82 |
+
data_dir=f"data/{lang}",
|
| 83 |
+
)
|
| 84 |
+
except Exception:
|
| 85 |
+
continue
|
| 86 |
+
for example in ds:
|
| 87 |
+
if count >= self.max_samples_per_language:
|
| 88 |
+
break
|
| 89 |
+
# Apply filters
|
| 90 |
+
stars = example.get("stars", 0) or 0
|
| 91 |
+
if stars < self.min_stars:
|
| 92 |
+
continue
|
| 93 |
+
license_ = (example.get("license") or "").lower()
|
| 94 |
+
if license_ and license_ not in self.license_filter:
|
| 95 |
+
continue
|
| 96 |
+
content = example.get("content", "")
|
| 97 |
+
if not content or len(content) < 50:
|
| 98 |
+
continue
|
| 99 |
+
yield {
|
| 100 |
+
"lang": lang,
|
| 101 |
+
"license": license_,
|
| 102 |
+
"repo": example.get("repository", ""),
|
| 103 |
+
"path": example.get("path", ""),
|
| 104 |
+
"size": example.get("size", len(content)),
|
| 105 |
+
"stars": stars,
|
| 106 |
+
"content": content,
|
| 107 |
+
}
|
| 108 |
+
count += 1
|
| 109 |
+
|
| 110 |
+
def collect(self, output_dir: Optional[str] = None) -> str:
|
| 111 |
+
"""Collect all samples and write to JSONL. Returns the output file path."""
|
| 112 |
+
output_dir = output_dir or self.cache_dir
|
| 113 |
+
os.makedirs(output_dir, exist_ok=True)
|
| 114 |
+
output_path = os.path.join(output_dir, "the_stack_v2.jsonl")
|
| 115 |
+
total = 0
|
| 116 |
+
with open(output_path, "w", encoding="utf-8") as f:
|
| 117 |
+
for sample in self:
|
| 118 |
+
f.write(json.dumps(sample, ensure_ascii=False) + "\n")
|
| 119 |
+
total += 1
|
| 120 |
+
print(f"[TheStackCollector] Collected {total} samples → {output_path}")
|
| 121 |
+
return output_path
|
| 122 |
+
|
| 123 |
+
def stats(self) -> Dict[str, int]:
|
| 124 |
+
"""Return per-language sample counts (calls collect if not yet run)."""
|
| 125 |
+
counts: Dict[str, int] = {lang: 0 for lang in self.languages}
|
| 126 |
+
for sample in self:
|
| 127 |
+
counts[sample["lang"]] = counts.get(sample["lang"], 0) + 1
|
| 128 |
+
return counts
|
| 129 |
+
|
| 130 |
+
|
| 131 |
+
__all__ = ["TheStackCollector", "SUPPORTED_LANGUAGES", "PERMISSIVE_LICENSES"]
|
nexus/data/collectors/wikipedia_collector.py
ADDED
|
@@ -0,0 +1,164 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
Wikipedia Collector - Thu thập dữ liệu từ Wikipedia
|
| 3 |
+
====================================================
|
| 4 |
+
"""
|
| 5 |
+
from __future__ import annotations
|
| 6 |
+
|
| 7 |
+
import logging
|
| 8 |
+
import urllib.request
|
| 9 |
+
import urllib.parse
|
| 10 |
+
import json
|
| 11 |
+
from typing import List, Dict, Optional, Iterator, Any
|
| 12 |
+
from dataclasses import dataclass
|
| 13 |
+
|
| 14 |
+
logger = logging.getLogger(__name__)
|
| 15 |
+
|
| 16 |
+
|
| 17 |
+
@dataclass
|
| 18 |
+
class WikiArticle:
|
| 19 |
+
"""Một Wikipedia article."""
|
| 20 |
+
title: str
|
| 21 |
+
content: str
|
| 22 |
+
url: str
|
| 23 |
+
language: str
|
| 24 |
+
categories: List[str]
|
| 25 |
+
|
| 26 |
+
|
| 27 |
+
class WikipediaCollector:
|
| 28 |
+
"""Collect articles từ Wikipedia API.
|
| 29 |
+
|
| 30 |
+
Supports Vietnamese (vi) and English (en) Wikipedia.
|
| 31 |
+
"""
|
| 32 |
+
|
| 33 |
+
BASE_URLS = {
|
| 34 |
+
"vi": "https://vi.wikipedia.org/w/api.php",
|
| 35 |
+
"en": "https://en.wikipedia.org/w/api.php",
|
| 36 |
+
}
|
| 37 |
+
|
| 38 |
+
RANDOM_TOPICS = {
|
| 39 |
+
"vi": [
|
| 40 |
+
"Trí tuệ nhân tạo", "Học máy", "Mạng nơ-ron nhân tạo",
|
| 41 |
+
"Python (ngôn ngữ lập trình)", "JavaScript", "Linux",
|
| 42 |
+
"Cơ sở dữ liệu", "Thuật toán", "Cấu trúc dữ liệu",
|
| 43 |
+
"Lập trình hướng đối tượng", "API", "JSON", "Git",
|
| 44 |
+
"Hệ điều hành", "Máy học sâu", "Xử lý ngôn ngữ tự nhiên",
|
| 45 |
+
"Học sâu", "Big data", "Điện toán đám mây",
|
| 46 |
+
],
|
| 47 |
+
"en": [
|
| 48 |
+
"Artificial intelligence", "Machine learning", "Neural network",
|
| 49 |
+
"Python (programming language)", "JavaScript", "Linux",
|
| 50 |
+
"Database", "Algorithm", "Data structure",
|
| 51 |
+
"Object-oriented programming", "API", "JSON", "Git",
|
| 52 |
+
"Operating system", "Deep learning", "Natural language processing",
|
| 53 |
+
"Big data", "Cloud computing", "Transformer (deep learning model)",
|
| 54 |
+
"Large language model", "GPT-4", "BERT (language model)",
|
| 55 |
+
],
|
| 56 |
+
}
|
| 57 |
+
|
| 58 |
+
def __init__(self, language: str = "vi"):
|
| 59 |
+
self.language = language
|
| 60 |
+
self.base_url = self.BASE_URLS.get(language, self.BASE_URLS["en"])
|
| 61 |
+
|
| 62 |
+
def get_article(self, title: str) -> Optional[WikiArticle]:
|
| 63 |
+
"""Lấy nội dung một Wikipedia article theo title."""
|
| 64 |
+
params = {
|
| 65 |
+
"action": "query",
|
| 66 |
+
"titles": title,
|
| 67 |
+
"prop": "extracts|categories",
|
| 68 |
+
"exintro": "false",
|
| 69 |
+
"explaintext": "true",
|
| 70 |
+
"cllimit": "10",
|
| 71 |
+
"format": "json",
|
| 72 |
+
"redirects": "1",
|
| 73 |
+
}
|
| 74 |
+
|
| 75 |
+
url = f"{self.base_url}?{urllib.parse.urlencode(params)}"
|
| 76 |
+
|
| 77 |
+
try:
|
| 78 |
+
req = urllib.request.Request(url, headers={"User-Agent": "NexusCoder-Collector/0.2"})
|
| 79 |
+
with urllib.request.urlopen(req, timeout=30) as response:
|
| 80 |
+
data = json.loads(response.read().decode())
|
| 81 |
+
|
| 82 |
+
pages = data.get("query", {}).get("pages", {})
|
| 83 |
+
if not pages:
|
| 84 |
+
return None
|
| 85 |
+
|
| 86 |
+
page = list(pages.values())[0]
|
| 87 |
+
if "missing" in page:
|
| 88 |
+
return None
|
| 89 |
+
|
| 90 |
+
content = page.get("extract", "")
|
| 91 |
+
if not content or len(content) < 100:
|
| 92 |
+
return None
|
| 93 |
+
|
| 94 |
+
categories = []
|
| 95 |
+
for cat in page.get("categories", []):
|
| 96 |
+
categories.append(cat["title"].replace("Category:", ""))
|
| 97 |
+
|
| 98 |
+
title_resolved = page.get("title", title)
|
| 99 |
+
url_resolved = f"https://{self.language}.wikipedia.org/wiki/{urllib.parse.quote(title_resolved.replace(' ', '_'))}"
|
| 100 |
+
|
| 101 |
+
return WikiArticle(
|
| 102 |
+
title=title_resolved,
|
| 103 |
+
content=content,
|
| 104 |
+
url=url_resolved,
|
| 105 |
+
language=self.language,
|
| 106 |
+
categories=categories,
|
| 107 |
+
)
|
| 108 |
+
except Exception as e:
|
| 109 |
+
logger.error(f"Wikipedia fetch failed for '{title}': {e}")
|
| 110 |
+
return None
|
| 111 |
+
|
| 112 |
+
def collect(
|
| 113 |
+
self,
|
| 114 |
+
topics: Optional[List[str]] = None,
|
| 115 |
+
max_per_topic: int = 1,
|
| 116 |
+
) -> Iterator[Dict[str, Any]]:
|
| 117 |
+
"""Collect articles, yield as text samples."""
|
| 118 |
+
topics = topics or self.RANDOM_TOPICS.get(self.language, self.RANDOM_TOPICS["en"])
|
| 119 |
+
|
| 120 |
+
for topic in topics:
|
| 121 |
+
article = self.get_article(topic)
|
| 122 |
+
if article:
|
| 123 |
+
yield {
|
| 124 |
+
"text": f"# {article.title}\n\n{article.content}",
|
| 125 |
+
"source": f"wikipedia_{self.language}",
|
| 126 |
+
"language": self.language,
|
| 127 |
+
"metadata": {
|
| 128 |
+
"title": article.title,
|
| 129 |
+
"url": article.url,
|
| 130 |
+
"categories": article.categories,
|
| 131 |
+
},
|
| 132 |
+
}
|
| 133 |
+
|
| 134 |
+
def collect_random(self, count: int = 100) -> Iterator[Dict[str, Any]]:
|
| 135 |
+
"""Collect random articles via Wikipedia API."""
|
| 136 |
+
params = {
|
| 137 |
+
"action": "query",
|
| 138 |
+
"list": "random",
|
| 139 |
+
"rnnamespace": "0", # Main namespace
|
| 140 |
+
"rnlimit": str(count),
|
| 141 |
+
"format": "json",
|
| 142 |
+
}
|
| 143 |
+
|
| 144 |
+
url = f"{self.base_url}?{urllib.parse.urlencode(params)}"
|
| 145 |
+
|
| 146 |
+
try:
|
| 147 |
+
req = urllib.request.Request(url, headers={"User-Agent": "NexusCoder-Collector/0.2"})
|
| 148 |
+
with urllib.request.urlopen(req, timeout=30) as response:
|
| 149 |
+
data = json.loads(response.read().decode())
|
| 150 |
+
|
| 151 |
+
for item in data.get("query", {}).get("random", []):
|
| 152 |
+
article = self.get_article(item["title"])
|
| 153 |
+
if article:
|
| 154 |
+
yield {
|
| 155 |
+
"text": f"# {article.title}\n\n{article.content}",
|
| 156 |
+
"source": f"wikipedia_{self.language}_random",
|
| 157 |
+
"language": self.language,
|
| 158 |
+
"metadata": {
|
| 159 |
+
"title": article.title,
|
| 160 |
+
"url": article.url,
|
| 161 |
+
},
|
| 162 |
+
}
|
| 163 |
+
except Exception as e:
|
| 164 |
+
logger.error(f"Wikipedia random failed: {e}")
|