diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000000000000000000000000000000000000..5b5556aa2a628522cc48a3cb4f7f0515e47692ef --- /dev/null +++ b/.gitignore @@ -0,0 +1,74 @@ +# Byte-compiled / optimized / DLL files +__pycache__/ +*.py[cod] +*$py.class + +# C extensions +*.so + +# Distribution / packaging +.Python +build/ +develop-eggs/ +dist/ +downloads/ +eggs/ +.eggs/ +lib/ +lib64/ +parts/ +sdist/ +var/ +wheels/ +*.egg-info/ +.installed.cfg +*.egg + +# PyInstaller +*.manifest +*.spec + +# Installer logs +pip-log.txt +pip-delete-this-directory.txt + +# Unit test / coverage reports +htmlcov/ +.tox/ +.coverage +.coverage.* +.cache +nosetests.xml +coverage.xml +*.cover +.pytest_cache/ + +# Jupyter Notebook +.ipynb_checkpoints + +# Environments +.env +.venv +env/ +venv/ +ENV/ + +# IDE +.idea/ +.vscode/ +*.swp +*.swo + +# OS +.DS_Store +Thumbs.db + +# Project specific +checkpoints/ +*.pt +*.pth +*.bin +*.safetensors +logs/ +*.log +nexus_coder-*.pt diff --git a/.python-version b/.python-version new file mode 100644 index 0000000000000000000000000000000000000000..28d9a01b1faabb66c0b96a3a32018cf52f6bc077 --- /dev/null +++ b/.python-version @@ -0,0 +1 @@ +3.12.13 diff --git a/ADVERTISEMENT.txt b/ADVERTISEMENT.txt new file mode 100644 index 0000000000000000000000000000000000000000..88388a6fe6a1875afbd24d4d84fb946e33445d2e --- /dev/null +++ b/ADVERTISEMENT.txt @@ -0,0 +1,162 @@ +============================================================================ + NEXUS CODER v0.4 — CYBERFORGE EDITION + AI Code & Security Engine — Open Source +============================================================================ + + Created by: Hieu Louis + GitHub: https://github.com/mhieuhonda/NexusCoder + Year: 2026 + License: NAL-1.0 (Attribution Required) + Version: 0.4.0 + Python: 3.12.13 + + +┌──────────────────────────────────────────────────────────────────────────┐ +│ │ +│ NEXUS CODER — SIÊU AI MÃ NGUỒN MỞ CHO CODE & BẢO MẬT │ +│ │ +│ • 423 tỷ tham số tổng, 39 tỷ tham số kích hoạt mỗi token │ +│ • Cửa sổ ngữ cảnh 3 TRIỆU tokens │ +│ • 60+ kỹ năng (skills) tích hợp │ +│ • 80+ công cụ (tools) tự động đăng ký │ +│ • Kiến trúc MoE Transformer thế hệ mới │ +│ │ +└──────────────────────────────────────────────────────────────────────────┘ + + +TẠI SAO NEXUS CODER KHÁC BIỆT? +============================== + +Nexus Coder v0.4 là một kiến trúc AI mã nguồn mở hoàn chỉnh, được Hieu Louis +thiết kế từ con số không. Repository này cung cấp: + + ✓ Toàn bộ mã nguồn kiến trúc model (Python/PyTorch) + ✓ Pipeline thu thập và xử lý dữ liệu code từ hàng nghìn GitHub repos + ✓ Framework huấn luyện đa giai đoạn + ✓ 60+ skills (code generation, debugging, security audit, ...) + ✓ 80+ tools (file ops, exec, web, database, devops, ...) + ✓ Tương thích Python 3.12.13 (strict) + + +TRUNG THỰC VỀ TRẠNG THÁI MODEL +================================ + + ⚠ REPO NÀY KHÔNG CHỨA MODEL ĐÃ ĐƯỢC TRAIN. + + Nexus Coder v0.4 phân phối MÃ NGUỒN của kiến trúc, pipeline dữ liệu, + và framework huấn luyện. Người dùng tự huấn luyện mô hình trên dữ + liệu của mình. Mọi thông tin quảng cáo về "performance" hay "benchmark" + chỉ là ước tính lý thuyết dựa trên kích thước kiến trúc — chưa có + model thực tế nào được train và đánh giá chính thức. + + Khi bạn thấy ai đó chia sẻ "Nexus Coder đã đạt X điểm benchmark Y", hãy + hỏi xem họ có train model thực tế hay không, và với dữ liệu gì. + + +TÍNH NĂNG KỸ THUẬT CHÍNH +========================== + +• Kiến trúc MoE Transformer với GQA (Grouped Query Attention) +• RoPE + YaRN scaling cho context window cực dài (3M tokens) +• FlashAttention-2 + SDPA + manual fallback +• Sliding Window Attention cho long-context efficiency +• QK-norm (Llama-3 style) cho training stability +• KV cache quantization (int8 / fp8) cho inference memory +• MLP-parallel (gate + up fuses thành 1 matmul) +• Gradient checkpointing cho training VRAM tiết kiệm +• Adaptive Density Routing (top-2 → top-8 experts theo input) +• 48 experts chuyên biệt hóa theo domain code (Python, JS, Rust, ...) + +• 8 nguồn dữ liệu: GitHub curated corpus (1000+ repos), HuggingFace, + arXiv, Wikipedia, StackOverflow, The-Stack v2, StarCoder2-data, + Python-Alpaca + +• Tích hợp 5 framework tham chiếu: litgpt, LlamaFactory, axolotl, + OpenHands, omp-gym (xem ATTRIBUTIONS.md) + + +CẤU HÌNH VARIANTS +================= + + tiny — 5M params (CPU demo) + small — 125M params (1 GPU) + medium — 1B params (4-8 GPU) + large — 10B params (32+ GPU) — backward-compat với v0.3 + xlarge — ~30B params (64+ GPU) + 30b — 30B/3B (64-128 GPU, H100 cluster) + 70b — ~70B/~12B (research only) + 423b — 423B/39B + 3M context (DEFAULT v0.4) — frontier scale + + +CÀI ĐẶT +======== + + git clone https://github.com/mhieuhonda/NexusCoder.git + cd NexusCoder + python3.12.13 -m venv venv + source venv/bin/activate + pip install -r requirements.txt + + +SỬ DỤNG +======== + + # Xem tóm tắt cấu hình + python -c "from nexus.config import print_config_summary; print_config_summary()" + + # Tiny demo + python scripts/train.py --config tiny --steps 100 + + # Train (cần GPU) + python scripts/train.py --config large --steps 5000 --use-amp + + # Chat với Nexus Agent + python scripts/chat.py + + +GIẤY PHÉP — NAL-1.0 (ATTRIBUTION REQUIRED) +========================================== + + Nexus Coder v0.4 được phát hành dưới giấy phép NexusCoder Attribution + License v1.0 (NAL-1.0). Bạn được phép: + + ✓ Sử dụng cho bất kỳ mục đích nào (commercial hoặc non-commercial) + ✓ Sửa đổi, phân phối, sublicense + ✓ Train, fine-tune, distill, quantize, ... + ✓ Build sản phẩm, dịch vụ, nghiên cứu trên nền Nexus Coder + + BẮT BUỘC: + + • Phải ghi danh tác giả gốc: "Hieu Louis" + • Phải kèm link: https://github.com/mhieuhonda/NexusCoder + • Trong model cards, README, UI, About pages, API responses, + research citations — bất cứ nơi nào hợp lý và thông dụng. + + Không được: + ✗ Xóa hoặc làm mờ attribution notices + ✗ Cầm quyền tác giả của người khác + ✗ Implement technical measures để erase embedded authorship + + Xem LICENSE để biết chi tiết đầy đủ. + + +TÁC GIẢ +======== + + Hieu Louis — 2026 + GitHub: https://github.com/mhieuhonda + Project: https://github.com/mhieuhonda/NexusCoder + License: NAL-1.0 (Attribution Required) + + +KẾT LUẬN +======== + + Nexus Coder v0.4 là một dự án mã nguồn mở cá nhân, được xây dựng với + tinh thần minh bạch: đây là CODE của kiến trúc AI, KHÔNG PHẢI model đã + train. Bạn có thể dùng nó để tự train model của mình, học hỏi kiến trúc, + hoặc build sản phẩm riêng. Hãy ghi danh Hieu Louis khi làm vậy. + + Cảm ơn bạn đã quan tâm đến Nexus Coder. + + — Hieu Louis, 2026 diff --git a/AGENTS.md b/AGENTS.md new file mode 100644 index 0000000000000000000000000000000000000000..8ed6d778916a0a763c1442be9acf97f99ff51b81 --- /dev/null +++ b/AGENTS.md @@ -0,0 +1,118 @@ +# AGENTS.md — Guide for AI Agents Contributing to Nexus Coder + +> **Read this first.** This file is written for AI agents who want to contribute +> to this repository. Humans can skip to `CONTRIBUTING.md`. + +## What Nexus Coder Is (and Is Not) + +Nexus Coder is an **open AI architecture** — source code, data pipeline, and a +training framework for a large Mixture-of-Experts (MoE) Transformer focused on +code generation and security analysis. + +**Important honesty note:** the model is **NOT pretrained**. There are no +checkpoints, no weights, no benchmarks. What exists is the architecture source +(60+ skills, 80+ tools, model layers, tokenizer, trainers, data collectors). +Your contribution helps turn this architecture into something that can actually +run and learn. + +If you expected a finished model, stop here. If you want to help build one from +the ground up, you are in the right place. + +## How to Contribute (Agent Workflow) + +### 1. Understand the layout + +``` +nexus/ +├── model/ # MoE Transformer layers (attention, MoE, RoPE, ...) +├── tokenizer/ # tokenizer implementation +├── training/ # trainer + dataset pipeline +├── inference/ # generation +├── agent/ # planner, router, memory, safety +├── skills/ # 60+ skills (auto-discovery) +├── tools/ # 80+ tools (auto-discovery) +├── data/ # collectors + processors +├── optim/ # quantization, LoRA, distillation, pruning +├── safety/ # filters, guardrails +├── eval/ # benchmarks, metrics +└── integrations/ # litgpt, LlamaFactory, axolotl, OpenHands, omp-gym +``` + +### 2. Pick an entry point + +Good first tasks for agents: + +- **Run the tiny demo on CPU**: `python scripts/train.py --config tiny --steps 100` + — verify it works, report output. +- **Test a skill**: import a skill from `nexus/skills/` and exercise it against + sample input. Report what works / what breaks. +- **Test a tool**: pick a tool in `nexus/tools/` and exercise it. +- **Review an integration**: read `nexus/integrations/*.py` and report whether + the adapted pattern is faithful to the upstream project. +- **Improve docs**: `docs/` always needs sharper explanations. +- **Add tests**: `tests/` is thin — any new test is valuable. + +### 3. Workflow + +1. **Open an issue first** describing what you want to do (unless it is a + trivial fix). This prevents wasted work. +2. **Fork** the repo (or ask a maintainer for collaborator access). +3. Create a **branch**, make your change, and open a **pull request**. +4. In your PR description, state clearly: + - what you changed, + - how you verified it (command + output), + - which config/skill/tool you exercised. + +### 4. Self-sufficiency + +You are expected to **figure things out yourself**: + +- Setup: `python3.12.13 -m venv venv && pip install -r requirements.txt` +- Small configs (`tiny`, `small`) run on **CPU** — no GPU needed. +- Larger configs need real hardware; if you have GPUs, train and report. +- If something is ambiguous, open an issue rather than guessing silently. + +## You May Train the Model — Freely + +You are **explicitly welcome and encouraged to train the model yourself**. The +NAL-1.0 license allows it for any purpose. There is no gate, no permission +request needed, and no fee. + +- **Small configs** (`tiny`, `small`) run on CPU — try a real training run + today with zero GPU cost. +- **Larger configs** (`medium`, `large`, `423b`) need serious hardware. If you + have GPUs (or a cloud budget), you are free to use them. +- Use the bundled data pipeline (`nexus/data/`) or your own data — both are fine. +- **Report your results honestly.** If you train something and get real + numbers, open a PR or an issue with your findings. The community (and the + project) wants to see real training logs, not hype. + +The project does not provide compute. You bring the hardware; we provide the +architecture, the data pipeline, and the training framework. + +> **Note for fairness:** training a model on this architecture does not make +> you a contributor to the repository. Attribution requirements still apply +> (NAL-1.0) whenever you use the trained result. To be listed as a contributor, +> submit a PR that improves the repo. + +## Attribution & License + +- License: **NAL-1.0** (Attribution Required). You may use/modify/distribute/ + train freely, but must credit the original author **Hieu Louis** + (github.com/mhieuhonda) wherever the work is used. +- Contributors are listed in `ATTRIBUTIONS.md`. If your PR merges, add yourself + there (or ask a maintainer to add you). + +## Communication + +- GitHub Issues + PRs are the coordination surface. +- For discussion, join the **code-realm** community on Moltbook (ask the + maintainer for the invite). The project has an active presence there. + +## Golden Rules + +1. **Be honest** — this is an untrained architecture. Never claim benchmark + results that do not exist. +2. **Small, verifiable PRs** beat big unverifiable ones. +3. **Reproduce before you report** — always run the thing you are claiming. +4. **Credit the author** in any downstream work (NAL-1.0). diff --git a/ATTRIBUTIONS.md b/ATTRIBUTIONS.md new file mode 100644 index 0000000000000000000000000000000000000000..128afdd06434b731778f56fc7a0dbd0f0bd08b3e --- /dev/null +++ b/ATTRIBUTIONS.md @@ -0,0 +1,114 @@ +# Attributions + +Nexus Coder v0.3 adapts ideas and code patterns from the following open-source projects. +All credit for the original algorithms goes to their respective authors. The code in +`nexus/integrations/` is rewritten to integrate cleanly into Nexus Coder's architecture; +it is NOT a vendored copy. + +## Reference Frameworks + +### 1. LitGPT (Lightning AI) +- **License**: Apache 2.0 +- **Source**: https://github.com/Lightning-AI/litgpt +- **What we adapted**: + - RoPE scaling strategies (linear / NTK-aware / YaRN) → `nexus/model/rope.py` + - FusedLinear pattern (concatenated Q/K/V projections) → `nexus/integrations/litgpt.py` + - PyTorch SDPA backend selection → `nexus/model/flash_attention.py` +- **Original attribution**: LitGPT: Lightning AI's LLM training toolkit. Authors: Karpathy et al. (Lightning AI), 2023-2024. + +### 2. LLaMA Factory (hiyouga) +- **License**: Apache 2.0 +- **Source**: https://github.com/hiyouga/LlamaFactory (also https://github.com/hiyouga/LLaMA-Factory) +- **What we adapted**: + - Dataset format converters (Alpaca / ShareGPT / ChatML / Completion → unified Nexus format) → `nexus/integrations/llamafactory.py` + - Concept of unified dataset registry → `nexus/data/collectors/` +- **Original attribution**: LlamaFactory: Unify Fine-tuning 100+ LLMs. Author: hiyouga. + +### 3. Axolotl (axolotl-ai-cloud) +- **License**: Apache 2.0 +- **Source**: https://github.com/axolotl-ai-cloud/axolotl +- **What we adapted**: + - AxolotlStyleConfig dataclass (typed training config schema) → `nexus/integrations/axolotl.py` + - Concept of single-YAML training configuration +- **Original attribution**: Axolotl: a simple tool for fine-tuning LLMs. Authors: winglian + axolotl-ai-cloud contributors. + +### 4. OpenHands +- **License**: MIT +- **Source**: https://github.com/OpenHands/OpenHands +- **What we adapted**: + - AgentLoop pattern (planner / executor / observer / reflector) → `nexus/integrations/openhands.py` + - Concept of structured agent loop with reflection +- **Original attribution**: OpenHands (formerly OpenDevin): an open platform for AI software developers. Authors: OpenHands contributors. + +### 5. omp-gym (Dylan Tirandaz) +- **License**: MIT +- **Source**: https://github.com/dylantirandaz/omp-gym +- **What we adapted**: + - OpenMP optimization benchmark tasks → `nexus/integrations/omp_gym.py` + - Concept of "predict-the-optimization" eval task +- **Original attribution**: omp-gym: An OpenMP optimization gym environment. Author: Dylan Tirandaz. + +## Other Attribution + +### Algorithms implemented in `nexus/model/` +- **RoPE**: Su et al., "RoFormer: Enhanced Transformer with Rotary Position Embedding" (2021). https://arxiv.org/abs/2104.09864 +- **FlashAttention**: Dao et al., "FlashAttention: Fast and Memory-Efficient Exact Attention with IO-Awareness" (2022). https://arxiv.org/abs/2205.14135 +- **FlashAttention-2**: Dao, "FlashAttention-2: Faster Attention with Better Parallelism and Work Partitioning" (2023). https://arxiv.org/abs/2307.08691 +- **ALiBi**: Press et al., "Train Short, Test Long: Attention with Linear Biases Enables Input Length Extrapolation" (ICLR 2022). https://arxiv.org/abs/2108.12409 +- **Sliding Window Attention**: Beltagy et al., "Longformer: The Long-Document Transformer" (2020). https://arxiv.org/abs/2004.05150 +- **YaRN**: Peng et al., "YaRN: Efficient Context Window Extension of Large Language Models" (2023). https://arxiv.org/abs/2309.00071 +- **NTK-aware RoPE scaling**: bloc97, "NTK-Aware Scaled RoPE" (2023). https://www.reddit.com/r/LocalLLaMA/comments/14lzrgj/ +- **SwiGLU**: Shazeer, "GLU Variants Improve Transformer" (2020). https://arxiv.org/abs/2002.05202 +- **RMSNorm**: Zhang & Sennrich, "Root Mean Square Layer Normalization" (2019). https://arxiv.org/abs/1910.07467 +- **GQA**: Ainslie et al., "GQA: Training Generalized Multi-Query Transformer Models from Multi-Head Checkpoints" (2023). https://arxiv.org/abs/2305.13245 +- **MoE**: Shazeer et al., "Outrageously Large Neural Networks: The Sparsely-Gated Mixture-of-Experts Layer" (2017). https://arxiv.org/abs/1701.06538 +- **Switch Transformer**: Fedus et al., "Switch Transformers: Scaling to Trillion Parameter Models with Simple and Efficient Sparsity" (2021). https://arxiv.org/abs/2101.03961 + +### Datasets referenced in `configs/sources.yaml` +- **The-Stack v2**: BigCode, https://huggingface.co/datasets/bigcode/the-stack-v2-train-full-ids +- **StarCoder2-data**: BigCode, https://huggingface.co/datasets/bigcode/starcoder2data +- **CodeParrot**: CodeParrot, https://huggingface.co/codeparrot +- **Wikipedia**: Wikimedia, https://huggingface.co/wikimedia/wikipedia +- **OSCAR**: https://oscar-project.org +- **UltraChat**: HuggingFaceH4, https://huggingface.co/HuggingFaceH4/ultrachat_200k +- **OpenHermes**: teknium, https://huggingface.co/teknium/OpenHermes-2.5 +- **OpenOrca**: https://huggingface.co/Open-Orca/OpenOrca +- **MetaMathQA**: https://huggingface.co/meta-math/MetaMathQA +- **GSM8K**: https://huggingface.co/datasets/gsm8k +- **HumanEval**: OpenAI, https://huggingface.co/datasets/openai_humaneval +- **MBPP**: Google Research, https://huggingface.co/datasets/mbpp +- **MATH**: https://huggingface.co/datasets/competition_math +- **FineWeb**: HuggingFaceFW, https://huggingface.co/datasets/HuggingFaceFW/fineweb +- **Open-Web-Math**: https://huggingface.co/datasets/open-web-math/open-web-math +- **Dolma**: AllenAI, https://huggingface.co/datasets/allenai/dolma +- **Pile**: EleutherAI, https://huggingface.co/datasets/EleutherAI/pile +- **C4**: Google, https://huggingface.co/datasets/c4 + +### Tools inspired by existing libraries +- The `Tool` and `Skill` base classes follow the OpenAI function-calling schema pattern +- Database tools wrap established client libraries (psycopg2, pymysql, redis, pymongo, etc.) +- Web tools use `requests` + `BeautifulSoup` conventions + +## License + +Nexus Coder is licensed under the MIT License (see [LICENSE](LICENSE)). + +The adaptations from the above projects comply with their respective licenses: +- Apache 2.0 components: retain notice, state changes +- MIT components: retain copyright notice + +Where algorithms are reimplemented from academic papers, the original papers +are cited in the source files. + +--- + +*This file is part of Nexus Coder v0.3 by Hieu Louis (2026).* + + +## Contributors + +> Maintained by hand. Add yourself here when your PR is merged, or ask a +> maintainer to add you. AI agents are welcome contributors. + +| Date | Contributor | Contribution | +|------|-------------|--------------| diff --git a/CHANGELOG.md b/CHANGELOG.md new file mode 100644 index 0000000000000000000000000000000000000000..847556a73958b5c9e1cfaf60ed85c46cbef0673b --- /dev/null +++ b/CHANGELOG.md @@ -0,0 +1,349 @@ +# Thay đổi / Changelog + +## v0.4.0 - 2026-08-17 — CyberForge Edition + +### SUPREME UPGRADE — 423B params, 3M context, CyberGym training methodology + +**Tác giả / Author**: Hieu Louis + +#### New Features + +##### Model architecture — 423B / 39B / 3M context +- New config `423b` (DEFAULT for v0.4): 423B total / 39B active params +- 24 layers, hidden 7168, 48 experts (4 active), inter 16384 +- 3,000,000-token context window via YaRN RoPE scaling (×60) +- Sliding window 32k + QK-norm + KV cache int8 + gradient checkpointing +- Adaptive Density Routing: top-2 → top-8 active experts based on input entropy + +##### CyberGym training methodology (NEW) +- **Code Genome Initialization (CGI)**: weight init from code motifs +- **Mutation Pressure Training (MPT)**: beneficial weight perturbations during training +- **Expert Speciation Curriculum (ESC)**: 48 experts → 48 species (Python/JS/Rust/Go/...) +- **Recursive Self-Compression (RSC)**: periodic self-distillation snapshots +- **Context Expansion Protocol (CEP)**: progressive 32k → 3M context extension +- **Adaptive Density Routing (ADR)**: entropy-based top-k routing +- Orchestrator `CyberForgeTrainer` wires all components together + +##### Data pipeline — Code corpus curated +- `configs/code_corpus.yaml`: 1000+ curated GitHub repos across 17 categories +- Categories: python_core, python_web, python_data, python_ml, python_dl, + python_tools, javascript_core, javascript_frameworks, rust_core, go_core, + java_core, c_cpp, devops, security, ai_tools, scientific, systems + +#### Bug Fixes (48 total) + +##### CRITICAL (6 fixes) +- `nexus/safety/__init__.py`: missing `get_default_guardrails` export broke `nexus.agent` +- `nexus/data/processors/deduplicator.py`: wrong import path (`.._logging_helpers` → `...utils.logging`) +- `nexus/model/attention.py`: INT8 KV cache quantization discarded scale → crash on 2nd decode step +- `scripts/collect_data.py`: `CURATED_TAGS` was a class attribute, not module-level → ImportError +- `nexus/agent/planner.py`: invalid dependency IDs silently treated as "met" (security bug) +- `nexus/config.py`: 30B / 70B configs were 5×–9× off their advertised size + +##### MAJOR (22 fixes) +- MoE never received `attention_mask` (padded tokens polluted aux loss) +- LoRA `target_modules` listed `gate_proj`/`up_proj` but v0.3 SwiGLU fuses them into `gate_up_proj` +- `python_exec` sandbox: when run as script, `__builtins__` was a module → sandbox escape +- `python_exec`: timeout was computed but never enforced → infinite loops could hang the agent +- `shell.py`: dead `if False` branch with unimported `os` +- ALiBi `max_slope` parameter was hardcoded to 8.0 (parameter had no effect) +- ALiBi non-power-of-2 head count subselection was wrong (took first N, not closest N) +- GitHub collector: `"c++"` language key didn't exist in EXTENSIONS (should be `"cpp"`) +- GitHub collector: hardcoded `--branch main` failed for repos using `master` +- arXiv collector: `.find().text` without None check crashed entire parse on missing element +- arXiv collector: query string not URL-encoded +- `compute_rouge`: rouge_1 was precision, not recall (corrected to F1) +- `compute_bleu`: empty references list crashed `min()` call +- FP8 quantization skip_layers comparison never matched (all params got FP8-quantized) +- Attention mask shape mismatch with KV cache + sliding window +- Trainer: AMP scaler state not checkpointed (resume caused NaN gradients) +- `quality_filter`: off-by-one in 10-gram repetition window +- `dataset.py`: hardcoded pad id 0 (collided with token 0 if user changed `pad_token_id`) +- Tokenizer: Vietnamese char `Ẵ` was duplicated as `Ẳ` (missing `Ẵ`) +- Tokenizer: BPE merge lost `` marker when first symbol had it +- Tokenizer: `tuple(k.split("|"))` broke when token contained `|` +- `scripts/train.py`: `--config` choices missing `30b`, `70b`, `423b` + +##### MINOR (20 fixes) +- Various unused imports, dead code, type hints +- See git log for full list + +#### License change +- Switched from MIT to **NexusCoder Attribution License v1.0 (NAL-1.0)** +- Free use for any purpose (commercial/non-commercial/research) +- Mandatory attribution: "Hieu Louis" + link to original repo +- See [LICENSE](LICENSE) for full terms + +#### Files added +- `nexus/cybergym/__init__.py` +- `nexus/cybergym/mutation.py` +- `nexus/cybergym/genome.py` +- `nexus/cybergym/adaptive_routing.py` +- `nexus/cybergym/speciation.py` +- `nexus/cybergym/compression.py` +- `nexus/cybergym/context_expansion.py` +- `nexus/cybergym/trainer.py` +- `configs/nexus_coder_423b.yaml` +- `configs/code_corpus.yaml` +- `ADVERTISEMENT.txt` + +--- + +## v0.3.0 - 2026-08-16 + +### 🚀 MASSIVE UPGRADE - Architecture + 4× Skills + 4× Tools + Massive Data + +**Tác giả / Author**: Hieu Louis + +#### ✨ Tính năng mới / New Features + +##### 🏗️ Kiến trúc v0.3 (NEW) +- ✅ **FlashAttention-2**: Optional `flash_attn` package backend (falls back to SDPA) +- ✅ **ALiBi position bias**: Alternative to RoPE for long-context extrapolation (Press et al., 2022) +- ✅ **Sliding Window Attention**: Alternating SWA / global layers (Longformer / Mistral style) +- ✅ **QK-norm**: RMSNorm on query/key for training stability (Llama-3 style) +- ✅ **MLP-parallel**: Fused gate+up projection (concatenated matmul) — faster on modern GPUs +- ✅ **KV cache quantization**: int8 / fp8 options for inference memory reduction +- ✅ **Gradient checkpointing**: Trade compute for VRAM at training time +- ✅ **RoPE scaling strategies**: linear / dynamic (NTK) / ntk / yarn — supports context extension up to 256k + +##### 📊 Multi-Variant Configs (7 variants) +- ✅ `tiny` - ~5M params (CPU demo) +- ✅ `small` - ~125M params (1 GPU) +- ✅ `medium` - ~1B params (4-8 GPU) +- ✅ `large` - 10B/1.5B (default, 32+ GPU) +- ✅ `xlarge` - ~30B/3B (research, 64+ GPU) +- ✅ `30b` - 30B/3B (v0.3 NEW, 64-128 H100, 64k context) +- ✅ `70b` - 70B/5B (v0.3 NEW, 256+ H100/H200, 128k context with YaRN ×4) + +##### 🎯 Skills System (15 → 60+) +- ✅ **Existing 15**: code_generation, code_review, code_refactor, debugging, documentation, testing, algorithm_design, data_analysis, translation, summarization, reasoning, math_skill, sql_generation, security_audit, performance_opt +- ✅ **DevOps (5 NEW)**: devops_skill, ci_cd_pipeline, release_management, monitoring, logging_analytics +- ✅ **ML (10 NEW)**: ml_training, ml_inference, ml_evaluation, ml_data_preprocessing, ml_feature_engineering, ml_hyperparameter_tuning, ml_model_explainability, ml_model_selection, ml_metrics, anomaly_detection +- ✅ **Data (5 NEW)**: data_pipeline, statistical_analysis, time_series_forecasting, clustering_analysis, knowledge_graph +- ✅ **Code (10 NEW)**: code_translation, code_completion, code_explanation, code_minification, code_documentation_generation, code_duplication_detection, code_dead_code_analysis, code_complexity_analysis, code_dependency_analysis, bug_reproduction +- ✅ **System (4 NEW)**: system_design, api_design, graphql_skill, microservices +- ✅ **Language (5 NEW)**: prompt_engineering, sentiment_analysis, topic_modeling, language_detection, creative_writing +- ✅ **Cloud (1 NEW)**: cloud_deploy +- ✅ **Blockchain (1 NEW)**: blockchain_audit +- ✅ **Caching (1 NEW)**: caching_strategy +- ✅ **Classification (1 NEW)**: classification_automation +- ✅ **Regex (1 NEW)**: regex_master +- ✅ **Shell (1 NEW)**: shell_scripting + +##### 🔧 Tools System (18+ → 80+) +- ✅ **Existing 24**: file_read/write/list/delete, shell_exec, python_exec, git_ops, http_request, web_fetch, web_search, code_search/lint/format, calculator, json/yaml/csv_parse, regex_search, archive, hash, encrypt, datetime, dns_lookup, ping +- ✅ **Database (12 NEW)**: sql_runner, sql_formatter, sql_migrator, postgres, mysql, sqlite, redis, mongo, elasticsearch, kafka, rabbitmq, graphql_client +- ✅ **DevOps/Cloud (12 NEW)**: docker, kubectl, terraform, ansible, aws_cli, gcloud_cli, azure_cli, ssh, scp, rsync, systemd, crontab +- ✅ **Code analysis (13 NEW)**: code_ast, code_complexity, code_dependency, code_metrics, code_smells, code_formatter_advanced, code_minifier, code_transpiler, code_runner, code_tester, code_compiler, code_profiler, code_coverage +- ✅ **Web/Network (12 NEW)**: websocket_client, grpc_client, url_shortener, dns_query, traceroute_tool, port_scanner, ssl_checker, ssl_generator, cert_checker, web_scraper, web_crawler, web_auth +- ✅ **Misc/Convert/Security (13 NEW)**: jwt_tool, oauth_tool, api_key_validator, markdown_converter, pdf_generator, image_processor, statistics_tool, linear_algebra_tool, probability_tool, ml_metrics_tool, model_evaluator, benchmark_runner, log_analyzer + +##### 📊 Data Pipeline (5 → 8 sources, 60 → 500+ repos) +- ✅ `GitHubCollector` (expanded): 60+ → 500+ curated repos (Python, JS, TS, Go, Rust, C/C++, Java, C#, Ruby, PHP, Swift, Kotlin, ...) +- ✅ `HuggingFaceCollector` (expanded): 20+ → 150+ datasets (code, instruction, math, Vietnamese, multilingual) +- ✅ `ArxivCollector`: 20 → 40 queries +- ✅ `WikipediaCollector`: 18 → 50+ topics per language +- ✅ `StackOverflowCollector`: 30 → 47 tags +- ✅ `TheStackCollector` (v0.3 NEW): BigCode's The-Stack v2 (~600 languages) +- ✅ `StarCoder2Collector` (v0.3 NEW): github_code + commits + jupyter notebooks +- ✅ `PythonAlpacaCollector` (v0.3 NEW): aggregates 6 Python instruction datasets + +##### 🧠 Processors (4 → 6) +- ✅ `TextCleaner`, `Deduplicator`, `QualityFilter`, `CodeFormatter` (existing) +- ✅ `LanguageIdProcessor` (v0.3 NEW): identifies vi/en/code, drops mislabeled +- ✅ `CodeQualityProcessor` (v0.3 NEW): scores Python 1-10 (docstring, type hints, no eval, etc.) + +##### 🤝 Integrations (5 reference frameworks) +- ✅ `litgpt.py`: FusedLinear adapter (Apache 2.0, Lightning AI) +- ✅ `llamafactory.py`: dataset format converters (alpaca/sharegpt/chatml/completion → nexus) +- ✅ `axolotl.py`: AxolotlStyleConfig dataclass (typed training config schema) +- ✅ `openhands.py`: AgentLoop pattern (planner/executor/observer/reflector) +- ✅ `omp_gym.py`: OpenMP optimization benchmark tasks + +##### 📈 Evaluation Module +- ✅ `BenchmarkSuite` - 10 benchmarks (HumanEval, MBPP, GSM8K, MMLU, BBH, MATH, ARC, TruthfulQA, AlpacaFarm, OMP-gym) +- ✅ Metrics: Perplexity, BLEU, ROUGE, F1, code-pass@k + +#### 🔧 Cải tiến / Improvements + +- ✅ **Auto-discovery registries**: Skills + Tools now scan directories dynamically — drop a `.py` file with a `Skill`/`Tool` subclass and it auto-registers +- ✅ **Stream-friendly training data**: `StreamingNexusDataset` for >1M example datasets (no RAM pressure) +- ✅ **Trimmed hardcoded data**: AUTHOR_TRAINING_DATA 150+ → 15 core examples (rest loaded from JSONL) +- ✅ **Lazy imports**: Faster startup; optional deps only imported when needed +- ✅ **Type hints**: Full typing throughout +- ✅ **Safety first**: All DANGEROUS/DESTRUCTIVE tools have `requires_confirmation=True` + `dry_run` support +- ✅ **Audit logging**: All tool calls logged to JSONL with timestamp, args, result, duration +- ✅ **Bilingual**: Vietnamese + English throughout + +#### 📊 Thông số kỹ thuật / Technical Specs + +| Thông số | v0.2 | v0.3 | +|----------|------|------| +| Version | 0.2.0 | 0.3.0 | +| Skills | 15 | 60+ | +| Tools | 18+ | 80+ | +| Data sources | 5 | 8 | +| Curated repos | 60+ | 500+ | +| Curated datasets | 20+ | 150+ | +| Configs | 5 | 7 | +| Reference frameworks | 0 | 5 | +| Attention backends | 1 (SDPA) | 3 (SDPA + FA2 + ALiBi) | +| Python version | 3.12.13 | 3.12.13 (strict) | +| PyTorch | >= 2.0 | >= 2.0 (>= 2.3 for 70b config) | + +#### 📁 Cấu trúc thư mục v0.3 (key changes) + +``` +NexusCoder/ +├── nexus/ +│ ├── __init__.py # v0.3.0 metadata +│ ├── config.py # + 30b/70b configs + attention features +│ ├── model/ +│ │ ├── attention.py # + FA2, ALiBi, SWA, QK-norm, KV quant +│ │ ├── rope.py # + NTK/YaRN scaling +│ │ ├── flash_attention.py # NEW +│ │ ├── alibi.py # NEW +│ │ ├── sliding_window.py # NEW +│ │ ├── layers.py # + MLP-parallel SwiGLU +│ │ └── transformer.py # + gradient checkpointing +│ ├── training/ +│ │ └── dataset.py # trimmed + StreamingNexusDataset +│ ├── skills/ # 60+ skills, auto-discovery registry +│ ├── tools/ # 80+ tools, auto-discovery registry +│ ├── data/ +│ │ ├── collectors/ # 8 collectors (3 NEW) +│ │ └── processors/ # 6 processors (2 NEW) +│ └── integrations/ # NEW: 5 reference framework adapters +├── configs/ +│ ├── nexus_coder_30b.yaml # NEW +│ ├── nexus_coder_70b.yaml # NEW +│ └── sources.yaml # expanded to 500+ repos, 150+ datasets +├── ATTRIBUTIONS.md # NEW +├── requirements.txt # + 30 new optional deps +├── pyproject.toml # v0.3.0 + extras groups +└── setup.py # v0.3.0 +``` + +#### 🚀 Migration từ v0.2 + +v0.3 backward compatible với v0.2: +- `NexusConfig()` vẫn hoạt động (default = large 10B) +- `NexusAgent()` vẫn hoạt động +- `AUTHOR_TRAINING_DATA` vẫn có (nhưng được tinh gọn) +- `scripts/train.py` vẫn hoạt động (nhưng có thêm config 30b, 70b) + +Breaking changes (minor): +- `nexus.skills.registry._auto_register_defaults` giờ dùng dynamic discovery thay vì hardcoded imports +- `nexus.tools.registry._auto_register_defaults` tương tự +- `AUTHOR_TRAINING_DATA` giảm từ 150+ xuống 15 mẫu (phần còn lại load từ `data/processed/*.jsonl`) + +#### 📦 Dependencies mới + +```bash +# Database tools +pip install sqlalchemy psycopg2-binary pymysql redis pymongo elasticsearch kafka-python pika + +# Web/Network tools +pip install aiohttp websockets grpcio beautifulsoup4 lxml + +# DevOps tools +pip install paramiko kubernetes docker + +# Media/Convert tools +pip install Pillow reportlab markdown + +# ML tools +pip install scikit-learn scipy transformers accelerate peft + +# Crypto +pip install pyjwt + +# GPU acceleration +pip install flash-attn --no-build-isolation + +# All at once +pip install -e ".[all]" +``` + +--- + +## v0.2.0 - 2026-08-16 + +### 🚀 Major Upgrade - Skills, Tools, và Data Pipeline + +**Tác giả / Author**: Hieu Louis + +#### ✨ Tính năng mới / New Features + +##### 🎯 Skills System (15 skills) +- ✅ `code_generation` - Sinh code từ mô tả (Python, JS, Go, Rust, SQL, ...) +- ✅ `code_review` - Review code: bugs, security, performance +- ✅ `code_refactor` - Tái cấu trúc code (extract, rename, patterns) +- ✅ `debugging` - Debug đa ngôn ngữ với 7-step protocol +- ✅ `documentation` - Sinh docstrings, README, API docs +- ✅ `testing` - Unit/integration/E2E/property/mutation tests +- ✅ `algorithm_design` - Thiết kế thuật toán, complexity analysis +- ✅ `data_analysis` - EDA, statistics, visualization +- ✅ `translation` - Dịch song ngữ Việt-Anh +- ✅ `summarization` - Extractive + abstractive summarization +- ✅ `reasoning` - CoT, ToT, ReAct, self-consistency +- ✅ `math_skill` - Algebra, calculus, linear algebra, statistics +- ✅ `sql_generation` - SQL cho 7 dialects (Postgres, MySQL, ...) +- ✅ `security_audit` - OWASP Top 10, SAST, dependency scan +- ✅ `performance_optimization` - Profiling, bottleneck, optimization + +##### 🔧 Tools System (15+ tools) +- ✅ `file_read` / `file_write` / `file_list` / `file_delete` - File operations +- ✅ `shell_exec` - Execute bash commands (sandboxed) +- ✅ `python_exec` - Execute Python code (restricted namespace) +- ✅ `git_ops` - Git commands với safety classification +- ✅ `http_request` - HTTP GET/POST/PUT/DELETE +- ✅ `web_fetch` - Fetch webpage, extract text +- ✅ `web_search` - Web search (Google/Bing/Brave API) +- ✅ `code_search` - Regex search trong code files +- ✅ `code_lint` / `code_format` - Lint & format code +- ✅ `calculator` - Safe math expression eval +- ✅ `json_parse` / `yaml_parse` / `csv_parse` - Data parsers +- ✅ `regex_search` - Regex search trong files +- ✅ `archive` - ZIP/TAR create/extract/list +- ✅ `hash` / `encrypt` - Hashing & AES-256-GCM encryption +- ✅ `datetime` - DateTime operations + timezone convert +- ✅ `dns_lookup` / `ping` - Network diagnostics + +##### 📊 Training Data Pipeline +- ✅ `GitHubCollector` - Thu thập code từ 60+ curated GitHub repos +- ✅ `HuggingFaceCollector` - 20+ curated HF datasets (code, text, Vietnamese) +- ✅ `ArxivCollector` - Scientific papers từ arXiv API +- ✅ `WikipediaCollector` - Vietnamese + English Wikipedia +- ✅ `StackOverflowCollector` - Q&A từ StackOverflow API +- ✅ `TextCleaner` - HTML stripping, unicode normalize, whitespace cleanup +- ✅ `CodeFormatter` - Format code samples, detect language +- ✅ `Deduplicator` - MinHash LSH for near-duplicate detection +- ✅ `QualityFilter` - Quality scoring (length, diversity, repetition) +- ✅ `CurriculumLearning` - 4-stage curriculum (easy → expert) + +--- + +## v0.1.0 - 2026-08-16 + +### 🎉 Initial Release - Foundation + +**Tác giả / Author**: Hieu Louis + +#### Thêm mới / Added + +- ✅ Kiến trúc **Mixture of Experts (MoE)** với 24 experts, 3 active mỗi token +- ✅ Tổng **10 tỷ tham số (10B)** với chỉ **1.5 tỷ tham số active (1.5B)** mỗi token +- ✅ **Cửa sổ ngữ cảnh 50,000 tokens** với RoPE +- ✅ **Grouped Query Attention (GQA)** - 16 heads, 4 KV heads +- ✅ **RMSNorm** + **SwiGLU** activation +- ✅ **BPE Tokenizer** song ngữ Việt-Anh +- ✅ **Training script** với AdamW + cosine LR schedule +- ✅ **Inference engine** với top-k, top-p, temperature sampling +- ✅ **AI Agent wrapper** (Nexus Agent) với quản lý hội thoại +- ✅ **Hardcoded author info** - model luôn nhớ được tạo bởi Hieu Louis +- ✅ **Test suite** đầy đủ +- ✅ **Song ngữ Việt-Anh** trong README và giao tiếp +- ✅ **MIT License** +- ✅ Tương thích **Python 3.12.13** diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md new file mode 100644 index 0000000000000000000000000000000000000000..4574dc9b6b04861cc3d8e0267313fd01a655a7f8 --- /dev/null +++ b/CONTRIBUTING.md @@ -0,0 +1,66 @@ +# Contributing to Nexus Coder + +Thanks for your interest in contributing! This project is an open AI +architecture in active development. Both humans and AI agents are welcome. + +> **AI agents:** read `AGENTS.md` first — it is written specifically for you. + +## Code of Conduct + +Be respectful. This project is built by a small team with limited resources. +Good-faith contributions are valued; trolling, spamming, or fake claims are not. + +## What We Need Help With + +1. **Running the small configs** — verify `tiny` / `small` train and run on CPU. +2. **Testing skills & tools** — exercise `nexus/skills/` and `nexus/tools/`. +3. **Reviewing integrations** — verify patterns adapted from upstream projects. +4. **Tests** — `tests/` is thin; add coverage for model layers, tokenizer, tools. +5. **Docs** — architecture docs always need improvement. +6. **Training experiments** — if you have GPUs, try a small real training run + and report honestly what you observed. + +## Getting Started + +```bash +git clone https://github.com/mhieuhonda/NexusCoder.git +cd NexusCoder +python3.12.13 -m venv venv +source venv/bin/activate +pip install -r requirements.txt +``` + +Python version is **3.12.13 (strict)**. Use `pyenv` or similar to match it. + +## Contribution Workflow + +1. **Open an issue first** describing what you plan to do (check for existing + ones to avoid duplication). +2. **Fork the repo** and create a branch. +3. Make your changes, keeping them **small and focused**. +4. **Verify** your change locally before opening a PR. +5. Open the **pull request** and describe what you did and how you verified it. + +## Style + +- Follow the existing code style in the file you are touching. +- Add or update tests for any new code. +- Keep commit messages clear and descriptive. + +## Labels + +- `good first issue` — beginner-friendly tasks (agents: start here) +- `help wanted` — tasks where maintainers explicitly want outside help +- `bug` — something is broken +- `enhancement` — new feature or improvement + +## License & Attribution + +Contributions are licensed under **NAL-1.0** (Attribution Required). By +contributing, you agree your changes are covered by this license and that the +original author **Hieu Louis** (github.com/mhieuhonda) retains attribution +requirements. See `LICENSE` and `ATTRIBUTIONS.md`. + +## Questions + +Open an issue, or reach out through the **code-realm** community on Moltbook. diff --git a/LICENSE b/LICENSE new file mode 100644 index 0000000000000000000000000000000000000000..9e21711e561fea56b05eb9995e321ef104e55b23 --- /dev/null +++ b/LICENSE @@ -0,0 +1,193 @@ +NexusCoder Attribution License v1.0 (NAL-1.0) +============================================== +Copyright (c) 2026 Hieu Louis (https://github.com/mhieuhonda) + +This license applies to the Nexus Coder project, including all source code, +configuration files, documentation, model architecture, training methodology, +and associated materials contained in this repository. + +By exercising any rights granted by this license, you accept and agree to be +bound by its terms and conditions. + +---------------------------------------------------------------------- + +1. DEFINITIONS + + "Project" means the Nexus Coder project, including all software, model + architecture code, training scripts, configurations, documentation, and + data pipeline code contained in this repository. + + "Author" means Hieu Louis, the original creator of the Project + (GitHub: https://github.com/mhieuhonda). + + "Derivative Work" means any work, model, software, or artifact that is + based on, derived from, or incorporates any part of the Project, including + but not limited to: + - Fine-tuned or modified versions of the Project + - Models trained using the Project's architecture or methodology + - Software that redistributes, modifies, or builds upon the Project + - Repackaged versions of the Project, in whole or in part + + "Attribution" means clearly and prominently crediting the Author as + the original creator of the Project, in the manner specified in + Section 3 below. + + "You" or "Your" means any person or entity exercising rights under + this license. + +---------------------------------------------------------------------- + +2. GRANTED RIGHTS + + Subject to the terms of this license, the Author grants You a worldwide, + royalty-free, non-exclusive, perpetual license to: + + (a) Use, copy, modify, merge, publish, distribute, sublicense, and/or + sell copies of the Project, in whole or in part. + + (b) Train, fine-tune, distill, prune, quantize, or otherwise create + Derivative Works based on the Project, for any commercial or + non-commercial purpose. + + (c) Use the Project's architecture, methodology, training pipeline, + code corpus, or any other component to build Your own products, + services, research, or any other work. + + (d) Distribute Derivative Works under any license You choose, provided + that You comply with the Attribution requirement (Section 3). + +---------------------------------------------------------------------- + +3. ATTRIBUTION REQUIREMENT (MANDATORY) + + You MUST attribute the Author (Hieu Louis) as the original creator of + the Project in all of the following circumstances: + + (a) REDISTRIBUTION: When You distribute, publish, or make available + the Project (or any Derivative Work), You must include: + - The Author's name: "Hieu Louis" + - A link to the original project: + https://github.com/mhieuhonda/NexusCoder + - A notice that the work is based on or derived from the Project + + (b) MODELS TRAINED USING THE PROJECT: If You train, fine-tune, or + otherwise create a model using the Project's architecture, + methodology, training pipeline, code, or any other component: + - You MUST include in the model card, README, documentation, + or any other accompanying material: + "Built using Nexus Coder by Hieu Louis + (https://github.com/mhieuhonda/NexusCoder)" + - This attribution MUST be visible to end users of the model, + including in API responses, UI, model cards, or download pages + where reasonable and customary. + + (c) PRODUCTS & SERVICES: If You build a product, service, or application + that uses the Project or any Derivative Work: + - You MUST include in the product's documentation, About page, + or credits section: "Powered by Nexus Coder by Hieu Louis" + - If the product has an "About" or "Credits" UI element, + the attribution must appear there. + + (d) RESEARCH PUBLICATIONS: If You publish research that used the + Project, You MUST cite: + Hieu Louis. "Nexus Coder: AI Code & Security Engine (CyberForge + Edition)." https://github.com/mhieuhonda/NexusCoder, 2026. + + (e) FORKED REPOSITORIES: If You fork the Project on GitHub or any + similar platform: + - You MUST keep the attribution in the README and LICENSE + - You MUST NOT claim to be the original author + - You MAY add Your own authorship for Your own contributions + +---------------------------------------------------------------------- + +4. ATTRIBUTION FORMAT + + The attribution must be clear, visible, and accessible to end users. + Acceptable formats include (but are not limited to): + + Short form (for UI, API responses, footers): + "Powered by Nexus Coder by Hieu Louis" + + Medium form (for README, docs): + "Built using Nexus Coder by Hieu Louis + (https://github.com/mhieuhonda/NexusCoder)" + + Full form (for model cards, academic publications): + "This work is based on Nexus Coder (v0.4.0, CyberForge Edition), + created by Hieu Louis (https://github.com/mhieuhonda/NexusCoder) + and licensed under NAL-1.0." + +---------------------------------------------------------------------- + +5. NO WARRANTIES + + The Project is provided "AS IS", without warranty of any kind, express + or implied, including but not limited to the warranties of + merchantability, fitness for a particular purpose, and non-infringement. + In no event shall the Author be liable for any claim, damages, or + other liability, whether in an action of contract, tort, or otherwise, + arising from, out of, or in connection with the Project or the use or + other dealings in the Project. + +---------------------------------------------------------------------- + +6. NO ENDORSEMENT + + You MUST NOT use the Author's name, the Project's name, or any + associated trademarks to imply endorsement of Your product, service, + or research without prior written permission from the Author. + +---------------------------------------------------------------------- + +7. NON-INTERFERENCE WITH ATTRIBUTION + + You MUST NOT remove, obscure, or alter any attribution notices + included in the Project. You MUST NOT implement technical measures + (e.g., watermark removal, fine-tuning that erases embedded authorship + information) that would have the effect of obscuring or removing the + Author's attribution. + +---------------------------------------------------------------------- + +8. TERMINATION + + Your rights under this license terminate automatically if You fail to + comply with any of its terms, especially the Attribution requirement + (Section 3). Upon termination, You must cease all use and distribution + of the Project and any Derivative Works, and destroy all copies in + Your possession or control. + +---------------------------------------------------------------------- + +9. VERSIONING + + This is version 1.0 of the NexusCoder Attribution License ("NAL-1.0"). + Future versions of the license, if any, will be designated by incrementing + the version number. The Author may release updated versions of this + license to address new use cases or clarify existing terms, but such + updates will not retroactively change the terms under which You received + the Project unless You explicitly choose to adopt the new version. + +---------------------------------------------------------------------- + +10. ENTIRE AGREEMENT + + This license constitutes the entire agreement between You and the + Author with respect to the Project. If any provision of this license + is held to be unenforceable, the remaining provisions shall remain + in full force and effect. + +---------------------------------------------------------------------- + +For questions or to request alternative licensing terms, contact: + + Hieu Louis + GitHub: https://github.com/mhieuhonda + Year: 2026 + +---------------------------------------------------------------------- + +By using, copying, modifying, distributing, or training on the Project, +You acknowledge that You have read, understood, and agree to be bound by +the terms of this NexusCoder Attribution License v1.0. diff --git a/README.md b/README.md new file mode 100644 index 0000000000000000000000000000000000000000..e8cca31a416ffac2e04dab7184ed6208b017d635 --- /dev/null +++ b/README.md @@ -0,0 +1,134 @@ +
+ +# 🧠 Nexus Coder + +### AI Code & Security Engine — CyberForge Edition + +**An open architecture for next‑generation code generation and security analysis** + +[![Python](https://img.shields.io/badge/Python-3.12.13-blue.svg)](https://www.python.org/) +[![PyTorch](https://img.shields.io/badge/PyTorch-2.0+-ee4c2c.svg)](https://pytorch.org/) +[![License: NAL-1.0](https://img.shields.io/badge/License-NAL--1.0-orange.svg)](LICENSE) +[![Status: In Development](https://img.shields.io/badge/Status-In%20Development-yellow.svg)]() +[![PRs Welcome](https://img.shields.io/badge/PRs-welcome-brightgreen.svg)]() +[![GitHub stars](https://img.shields.io/github/stars/mhieuhonda/NexusCoder?style=social)](https://github.com/mhieuhonda/NexusCoder) +[![GitHub forks](https://img.shields.io/github/forks/mhieuhonda/NexusCoder?style=social)](https://github.com/mhieuhonda/NexusCoder) +[![GitHub last commit](https://img.shields.io/github/last-commit/mhieuhonda/NexusCoder)](https://github.com/mhieuhonda/NexusCoder) + +**Created by [Hieu Louis](https://github.com/mhieuhonda)** · 2026 + +
+ +## 📖 Introduction + +**Nexus Coder** is an open‑source AI architecture, designed from the ground up by **Hieu Louis**, focused on two core capabilities: + +- **High‑quality code generation** powered by a large‑scale Mixture‑of‑Experts (MoE) Transformer. +- **Deep security analysis** for source code and systems. + +The project is under **active development**. This repository provides: + +- The complete **model architecture source code** (Python/PyTorch). +- A **data collection and processing pipeline** for code from multiple sources. +- A **multi‑stage training framework** designed to scale. +- **60+ skills** and **80+ tools** with automatic registration. +- Configurations ranging from `tiny` (5M) to `423b` (423B parameters). + +> **Important:** The model is **not pretrained** yet. We distribute only the architecture source and training pipeline. Users need to train their own models on their own data, in compliance with the NAL‑1.0 license. + +## 📊 Key Technical Specifications + +| Item | Value | +|------|-------| +| Total parameters | ~423B | +| Active parameters per token | ~39B | +| Context window | 3,000,000 tokens (3M) | +| Architecture | MoE Transformer (GQA + RoPE/YaRN + RMSNorm + SwiGLU + FlashAttention‑2 + Sliding Window + QK‑norm + KV cache quantization + MLP‑parallel + Gradient checkpointing) | +| Skills | 60+ (code, devops, ML, data, security, cloud, system, blockchain, language) | +| Tools | 80+ (file, exec, web, code analysis, database, devops, crypto, math, network) | +| Data sources | 8+ (GitHub curated corpus, HuggingFace, arXiv, Wikipedia, StackOverflow, The‑Stack v2, StarCoder2‑data, Python‑Alpaca) | +| Python version | 3.12.13 (strict) | + +## 🚀 Quick Install + +```bash +git clone https://github.com/mhieuhonda/NexusCoder.git +cd NexusCoder +python3.12.13 -m venv venv +source venv/bin/activate +pip install -r requirements.txt +# or: pip install -e ".[all]" +``` + +💻 Usage + +```bash +# Print configuration summary +python -c "from nexus.config import print_config_summary; print_config_summary()" + +# Tiny demo (CPU) +python scripts/train.py --config tiny --steps 100 + +# Train larger configurations (requires GPU) +python scripts/train.py --config large --steps 5000 --use-amp +python scripts/train.py --config 423b --steps 50000 --use-amp --deepspeed +``` + +📁 Project Structure + +``` +NexusCoder/ +├── nexus/ # Main package +│ ├── model/ # MoE Transformer (attention, MoE, layers, ...) +│ ├── tokenizer/ +│ ├── training/ # Trainer + Dataset +│ ├── inference/ +│ ├── agent/ # Planner, Router, Memory, Safety +│ ├── skills/ # 60+ skills (auto‑discovery) +│ ├── tools/ # 80+ tools (auto‑discovery) +│ ├── data/ # Collectors + Processors +│ ├── optim/ # Quantize, LoRA, Distill, Prune +│ ├── safety/ # Filters, Guardrails +│ ├── eval/ # Benchmarks, Metrics +│ ├── integrations/ # litgpt, LlamaFactory, axolotl, OpenHands, omp‑gym +│ └── utils/ +├── configs/ # YAML configs (tiny → 423B) +├── scripts/ # CLI scripts +├── docs/ # ARCHITECTURE, TRAINING, SKILLS, TOOLS, DATA +├── tests/ +├── ATTRIBUTIONS.md +├── CHANGELOG.md +├── LICENSE # NAL‑1.0 (Attribution Required) +├── requirements.txt +├── pyproject.toml +├── setup.py +└── README.md +``` + +⚖️ License + +Released under the NexusCoder Attribution License v1.0 (NAL‑1.0). + +· You may use, modify, distribute, and train models for any purpose. +· Attribution is required to the original author: Hieu Louis (github.com/mhieuhonda). +· No warranty. See LICENSE for details. + +👤 Author + +
+ +Hieu Louis · 2026 + +· GitHub: @mhieuhonda +· Project: NexusCoder +· License: NAL‑1.0 (Attribution Required) + +
+ +
+ +Nexus Coder — CyberForge Edition + +Made by Hieu Louis · 2026 + +
diff --git a/configs/code_corpus.yaml b/configs/code_corpus.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0d1f6748d611b586796003af5614a4eccceb663f --- /dev/null +++ b/configs/code_corpus.yaml @@ -0,0 +1,5346 @@ +# ============================================================================ +# Nexus Coder v0.4 — Code Corpus (curated GitHub repos) +# ============================================================================ +# Curated list of high-quality open-source GitHub repositories used as +# training data for the CyberForge training pipeline (Code Genome Init + +# Expert Speciation Curriculum). +# +# Each repo: (owner, name, languages, max_files) +# Languages: list of primary languages (e.g. ["python", "c"]) +# max_files: cap on number of files to extract from this repo +# +# Author: Hieu Louis (2026) +# ============================================================================ + +version: 0.4.0 +total_repos: 1030 +last_updated: '2026-08-17' +categories: + python_core: + description: Core Python language, stdlib, tooling + repos: + - owner: python + name: cpython + languages: + - python + - c + max_files: 5000 + - owner: pypa + name: pip + languages: + - python + max_files: 2000 + - owner: pypa + name: setuptools + languages: + - python + max_files: 1500 + - owner: pypa + name: wheel + languages: + - python + max_files: 500 + - owner: pypa + name: build + languages: + - python + max_files: 500 + - owner: pypa + name: virtualenv + languages: + - python + max_files: 1500 + - owner: pypa + name: packaging + languages: + - python + max_files: 500 + - owner: python + name: peps + languages: + - python + max_files: 1000 + - owner: python + name: typeshed + languages: + - python + max_files: 5000 + - owner: python + name: mypy + languages: + - python + max_files: 3000 + - owner: python + name: pyperformance + languages: + - python + max_files: 500 + - owner: psf + name: requests + languages: + - python + max_files: 1000 + - owner: psf + name: black + languages: + - python + max_files: 2000 + - owner: pycqa + name: flake8 + languages: + - python + max_files: 1000 + - owner: pycqa + name: isort + languages: + - python + max_files: 500 + - owner: pycqa + name: pytest + languages: + - python + max_files: 3000 + - owner: pycqa + name: pylint + languages: + - python + max_files: 3000 + - owner: pycqa + name: bandit + languages: + - python + max_files: 500 + - owner: pycqa + name: coveragepy + languages: + - python + max_files: 2000 + - owner: pycqa + name: astroid + languages: + - python + max_files: 1500 + - owner: python-attrs + name: attrs + languages: + - python + max_files: 800 + - owner: pydantic + name: pydantic + languages: + - python + max_files: 2000 + - owner: encode + name: httpx + languages: + - python + max_files: 1500 + - owner: encode + name: starlette + languages: + - python + max_files: 1000 + - owner: encode + name: uvicorn + languages: + - python + max_files: 800 + python_web: + description: Python web frameworks (Django, Flask, ...) + repos: + - owner: django + name: django + languages: + - python + max_files: 5000 + - owner: pallets + name: flask + languages: + - python + max_files: 2000 + - owner: pallets + name: werkzeug + languages: + - python + max_files: 1000 + - owner: pallets + name: jinja + languages: + - python + max_files: 1000 + - owner: pallets + name: click + languages: + - python + max_files: 800 + - owner: pallets + name: itsdangerous + languages: + - python + max_files: 300 + - owner: pallets + name: markupsafe + languages: + - python + - c + max_files: 500 + - owner: tiangolo + name: fastapi + languages: + - python + max_files: 2000 + - owner: sanic-org + name: sanic + languages: + - python + max_files: 2000 + - owner: tornadoweb + name: tornado + languages: + - python + max_files: 3000 + - owner: pyramid + name: pyramid + languages: + - python + max_files: 1500 + - owner: bottlepy + name: bottle + languages: + - python + max_files: 800 + - owner: falconry + name: falcon + languages: + - python + max_files: 1500 + - owner: django + name: djangorestframework + languages: + - python + max_files: 3000 + - owner: wagtail + name: wagtail + languages: + - python + max_files: 5000 + - owner: mezzanine + name: mezzanine + languages: + - python + max_files: 2000 + - owner: viewflow + name: viewflow + languages: + - python + max_files: 1000 + - owner: python-restx + name: flask-restx + languages: + - python + max_files: 800 + - owner: flask-admin + name: flask-admin + languages: + - python + max_files: 1000 + - owner: flask-login + name: flask-login + languages: + - python + max_files: 500 + - owner: flask-cors + name: flask-cors + languages: + - python + max_files: 300 + - owner: marshmallow-code + name: marshmallow + languages: + - python + max_files: 2000 + - owner: sloria + name: textblob + languages: + - python + max_files: 800 + - owner: zopefoundation + name: zope + languages: + - python + max_files: 1500 + - owner: elastic + name: elasticsearch-py + languages: + - python + max_files: 1000 + - owner: andymccurdy + name: redis-py + languages: + - python + max_files: 1500 + - owner: mongodb + name: mongo-python-driver + languages: + - python + max_files: 3000 + - owner: sqlalchemy + name: sqlalchemy + languages: + - python + max_files: 5000 + - owner: coleifer + name: peewee + languages: + - python + max_files: 1500 + - owner: aws + name: chalice + languages: + - python + max_files: 1000 + - owner: graphql-python + name: graphene + languages: + - python + max_files: 2000 + - owner: graphql-python + name: graphql-core + languages: + - python + max_files: 1500 + - owner: strawberry-graphql + name: strawberry + languages: + - python + max_files: 2000 + - owner: ariadne + name: ariadne + languages: + - python + max_files: 1000 + - owner: celery + name: celery + languages: + - python + max_files: 4000 + - owner: celery + name: kombu + languages: + - python + max_files: 2000 + - owner: celery + name: vine + languages: + - python + max_files: 500 + - owner: python-rq + name: rq + languages: + - python + max_files: 1000 + - owner: profx + name: rq + languages: + - python + max_files: 1000 + - owner: apache + name: airflow + languages: + - python + max_files: 8000 + - owner: spotify + name: luigi + languages: + - python + max_files: 2000 + - owner: joblib + name: joblib + languages: + - python + max_files: 1500 + - owner: jupyter + name: notebook + languages: + - python + max_files: 5000 + - owner: jupyter + name: jupyterlab + languages: + - python + - typescript + max_files: 5000 + - owner: jupyter + name: ipython + languages: + - python + max_files: 4000 + - owner: jupyter + name: nbformat + languages: + - python + max_files: 500 + - owner: jupyter + name: nbconvert + languages: + - python + max_files: 1500 + - owner: jupyter-server + name: jupyter_server + languages: + - python + max_files: 1500 + - owner: ipython + name: ipykernel + languages: + - python + max_files: 500 + - owner: jupyter + name: qtconsole + languages: + - python + max_files: 800 + - owner: jupyter-widgets + name: ipywidgets + languages: + - python + max_files: 1500 + python_data: + description: Python data science (pandas, numpy, polars) + repos: + - owner: pandas-dev + name: pandas + languages: + - python + - c + max_files: 8000 + - owner: numpy + name: numpy + languages: + - python + - c + max_files: 5000 + - owner: scipy + name: scipy + languages: + - python + - c + - fortran + max_files: 5000 + - owner: pola-rs + name: polars + languages: + - python + - rust + max_files: 3000 + - owner: dask + name: dask + languages: + - python + max_files: 5000 + - owner: apache + name: arrow + languages: + - python + - c++ + max_files: 8000 + - owner: modin-project + name: modin + languages: + - python + max_files: 2000 + - owner: vaexio + name: vaex + languages: + - python + max_files: 2000 + - owner: pydata + name: xarray + languages: + - python + max_files: 3000 + - owner: numba + name: numba + languages: + - python + - c + max_files: 4000 + - owner: cupy + name: cupy + languages: + - python + - c++ + max_files: 4000 + - owner: h5py + name: h5py + languages: + - python + - c + max_files: 2000 + - owner: pytables + name: pytables + languages: + - python + - c + max_files: 1500 + - owner: zarr-developers + name: zarr + languages: + - python + max_files: 1500 + - owner: intake + name: intake + languages: + - python + max_files: 800 + - owner: glue-viz + name: glue + languages: + - python + max_files: 1500 + - owner: google + name: jax + languages: + - python + max_files: 8000 + - owner: arrayfire + name: arrayfire + languages: + - python + - c++ + max_files: 1500 + - owner: numexpr + name: numexpr + languages: + - python + - c + max_files: 800 + - owner: bloomberg + name: bqplot + languages: + - python + max_files: 1000 + - owner: bokeh + name: bokeh + languages: + - python + max_files: 5000 + - owner: matplotlib + name: matplotlib + languages: + - python + max_files: 8000 + - owner: plotly + name: plotly.py + languages: + - python + max_files: 5000 + - owner: altair-viz + name: altair + languages: + - python + max_files: 2000 + - owner: seaborn + name: seaborn + languages: + - python + max_files: 2000 + - owner: pyvista + name: pyvista + languages: + - python + max_files: 2000 + - owner: datashader + name: datashader + languages: + - python + max_files: 1500 + - owner: holoviz + name: holoviews + languages: + - python + max_files: 2500 + - owner: panel + name: panel + languages: + - python + max_files: 2000 + - owner: vega + name: vega-lite + languages: + - python + max_files: 1000 + python_ml: + description: Python ML (scikit-learn, xgboost, ...) + repos: + - owner: scikit-learn + name: scikit-learn + languages: + - python + - c + max_files: 10000 + - owner: dmlc + name: xgboost + languages: + - python + - c++ + max_files: 3000 + - owner: microsoft + name: LightGBM + languages: + - python + - c++ + max_files: 2500 + - owner: catboost + name: catboost + languages: + - python + - c++ + max_files: 3000 + - owner: scikit-learn-contrib + name: imbalanced-learn + languages: + - python + max_files: 1500 + - owner: scikit-learn-contrib + name: category-encoders + languages: + - python + max_files: 1000 + - owner: scikit-learn-contrib + name: hmmlearn + languages: + - python + max_files: 800 + - owner: scikit-learn-contrib + name: metric-learn + languages: + - python + max_files: 800 + - owner: scikit-learn-contrib + name: scikit-learn-extra + languages: + - python + max_files: 1000 + - owner: scikit-optimize + name: scikit-optimize + languages: + - python + max_files: 1000 + - owner: hyperopt + name: hyperopt + languages: + - python + max_files: 1000 + - owner: optuna + name: optuna + languages: + - python + max_files: 3000 + - owner: ray-project + name: ray + languages: + - python + - c++ + max_files: 8000 + - owner: interpretml + name: interpret + languages: + - python + max_files: 1500 + - owner: shap + name: shap + languages: + - python + - c++ + max_files: 2000 + - owner: lime-ml + name: lime + languages: + - python + max_files: 800 + - owner: pycaret + name: pycaret + languages: + - python + max_files: 2500 + - owner: alibaba + name: EasyNLP + languages: + - python + max_files: 2000 + - owner: PyTorchLightning + name: pytorch-lightning + languages: + - python + max_files: 4000 + - owner: PyTorchLightning + name: lightning-flash + languages: + - python + max_files: 1500 + - owner: huggingface + name: accelerate + languages: + - python + max_files: 2000 + - owner: huggingface + name: tokenizers + languages: + - python + - rust + max_files: 2000 + - owner: huggingface + name: datasets + languages: + - python + max_files: 3000 + - owner: huggingface + name: evaluate + languages: + - python + max_files: 800 + - owner: huggingface + name: peft + languages: + - python + max_files: 1000 + - owner: huggingface + name: transformers + languages: + - python + max_files: 12000 + - owner: explosion + name: spaCy + languages: + - python + - cython + max_files: 8000 + - owner: explosion + name: thinc + languages: + - python + - cython + max_files: 2000 + - owner: explosion + name: cymem + languages: + - python + - c + max_files: 300 + - owner: explosion + name: preshed + languages: + - python + - c + max_files: 300 + - owner: explosion + name: srsly + languages: + - python + - c + max_files: 300 + - owner: explosion + name: murmurhash + languages: + - python + - c + max_files: 300 + - owner: explosion + name: catalogue + languages: + - python + max_files: 200 + - owner: explosion + name: confection + languages: + - python + max_files: 200 + - owner: nltk + name: nltk + languages: + - python + max_files: 3000 + - owner: RasaHQ + name: rasa + languages: + - python + max_files: 5000 + - owner: RasaHQ + name: rasa-sdk + languages: + - python + max_files: 800 + - owner: cltk + name: cltk + languages: + - python + max_files: 1500 + - owner: stanfordnlp + name: stanza + languages: + - python + max_files: 2500 + - owner: stanfordnlp + name: CoreNLP + languages: + - java + max_files: 3000 + - owner: allenai + name: allennlp + languages: + - python + max_files: 4000 + - owner: allenai + name: allennlp-models + languages: + - python + max_files: 1500 + - owner: facebookresearch + name: fairseq + languages: + - python + max_files: 5000 + - owner: facebookresearch + name: ParlAI + languages: + - python + max_files: 3000 + - owner: facebookresearch + name: DrQA + languages: + - python + max_files: 1000 + - owner: facebookresearch + name: fastText + languages: + - python + - c++ + max_files: 2500 + - owner: facebookresearch + name: LASER + languages: + - python + max_files: 1000 + - owner: facebookresearch + name: XLM + languages: + - python + max_files: 1500 + - owner: facebookresearch + name: UnsupervisedMT + languages: + - python + max_files: 800 + - owner: facebookresearch + name: MUSE + languages: + - python + max_files: 1000 + - owner: google-research + name: bert + languages: + - python + max_files: 1000 + - owner: google-research + name: albert + languages: + - python + max_files: 500 + - owner: google-research + name: electra + languages: + - python + max_files: 500 + - owner: google-research + name: t5 + languages: + - python + max_files: 1500 + - owner: google-research + name: vision_transformer + languages: + - python + max_files: 500 + - owner: google-research + name: scenic + languages: + - python + max_files: 1000 + - owner: openai + name: gpt-2 + languages: + - python + max_files: 1500 + - owner: openai + name: whisper + languages: + - python + max_files: 1000 + - owner: openai + name: tiktoken + languages: + - python + - rust + max_files: 800 + - owner: openai + name: evals + languages: + - python + max_files: 1500 + - owner: openai + name: openai-python + languages: + - python + max_files: 1500 + - owner: microsoft + name: DeepSpeed + languages: + - python + - c++ + max_files: 3000 + - owner: microsoft + name: Megatron-DeepSpeed + languages: + - python + max_files: 2000 + - owner: microsoft + name: DeBERTa + languages: + - python + max_files: 1000 + - owner: microsoft + name: unilm + languages: + - python + max_files: 2000 + - owner: microsoft + name: nni + languages: + - python + max_files: 3000 + - owner: microsoft + name: CodeBERT + languages: + - python + max_files: 500 + - owner: microsoft + name: graphcodebert + languages: + - python + max_files: 500 + - owner: EleutherAI + name: gpt-neo + languages: + - python + max_files: 2000 + - owner: EleutherAI + name: gpt-j + languages: + - python + max_files: 1500 + - owner: EleutherAI + name: gpt-neox + languages: + - python + max_files: 2500 + - owner: EleutherAI + name: lm-evaluation-harness + languages: + - python + max_files: 1500 + - owner: bigscience-workshop + name: t-zero + languages: + - python + max_files: 1000 + - owner: bigscience-workshop + name: data_tooling + languages: + - python + max_files: 800 + python_dl: + description: Python deep learning (torch, tf, jax) + repos: + - owner: pytorch + name: pytorch + languages: + - python + - c++ + max_files: 15000 + - owner: pytorch + name: vision + languages: + - python + max_files: 3000 + - owner: pytorch + name: audio + languages: + - python + - c++ + max_files: 2000 + - owner: pytorch + name: text + languages: + - python + max_files: 2000 + - owner: pytorch + name: serve + languages: + - python + max_files: 1500 + - owner: pytorch + name: ignite + languages: + - python + max_files: 1500 + - owner: pytorch + name: captum + languages: + - python + max_files: 1500 + - owner: pytorch + name: fairseq + languages: + - python + max_files: 4000 + - owner: pytorch + name: examples + languages: + - python + max_files: 2000 + - owner: pytorch + name: xla + languages: + - python + - c++ + max_files: 2000 + - owner: tensorflow + name: tensorflow + languages: + - python + - c++ + max_files: 15000 + - owner: tensorflow + name: tensorboard + languages: + - python + max_files: 3000 + - owner: tensorflow + name: datasets + languages: + - python + max_files: 2000 + - owner: tensorflow + name: agents + languages: + - python + max_files: 1500 + - owner: tensorflow + name: probability + languages: + - python + max_files: 2000 + - owner: tensorflow + name: addons + languages: + - python + - c++ + max_files: 1500 + - owner: tensorflow + name: models + languages: + - python + max_files: 5000 + - owner: keras-team + name: keras + languages: + - python + max_files: 5000 + - owner: keras-team + name: keras-applications + languages: + - python + max_files: 1000 + - owner: keras-team + name: keras-tuner + languages: + - python + max_files: 1500 + - owner: keras-team + name: autokeras + languages: + - python + max_files: 2000 + - owner: google + name: jax + languages: + - python + max_files: 8000 + - owner: google + name: flax + languages: + - python + max_files: 2500 + - owner: google + name: optax + languages: + - python + max_files: 1000 + - owner: google + name: haiku + languages: + - python + max_files: 1000 + - owner: microsoft + name: CNTK + languages: + - python + - c++ + max_files: 3000 + - owner: apple + name: mlx + languages: + - python + - c++ + max_files: 2000 + - owner: openai + name: CLIP + languages: + - python + max_files: 1000 + - owner: openai + name: consistency_models + languages: + - python + max_files: 500 + - owner: openai + name: improved-diffusion + languages: + - python + max_files: 800 + - owner: openai + name: guided-diffusion + languages: + - python + max_files: 1000 + - owner: CompVis + name: stable-diffusion + languages: + - python + max_files: 3000 + - owner: CompVis + name: taming-transformers + languages: + - python + max_files: 1500 + - owner: CompVis + name: latent-diffusion + languages: + - python + max_files: 1500 + - owner: stabilityai + name: generative-models + languages: + - python + max_files: 2000 + - owner: stabilityai + name: stable-diffusion-3 + languages: + - python + max_files: 1500 + - owner: huggingface + name: diffusers + languages: + - python + max_files: 4000 + - owner: huggingface + name: safetensors + languages: + - python + - rust + max_files: 800 + - owner: huggingface + name: trl + languages: + - python + max_files: 1500 + - owner: huggingface + name: alignment-handbook + languages: + - python + max_files: 1000 + - owner: huggingface + name: optimum + languages: + - python + max_files: 1500 + - owner: huggingface + name: text-generation-inference + languages: + - python + - rust + max_files: 2500 + - owner: facebookresearch + name: segment-anything + languages: + - python + max_files: 1500 + - owner: facebookresearch + name: detectron2 + languages: + - python + max_files: 4000 + - owner: facebookresearch + name: pytorch3d + languages: + - python + max_files: 3000 + - owner: facebookresearch + name: metaseq + languages: + - python + max_files: 2000 + - owner: facebookresearch + name: xformers + languages: + - python + - cuda + max_files: 2000 + - owner: facebookresearch + name: hydra + languages: + - python + max_files: 2000 + - owner: facebookresearch + name: fvcore + languages: + - python + max_files: 1000 + - owner: facebookresearch + name: slowfast + languages: + - python + max_files: 1500 + - owner: facebookresearch + name: ClassyVision + languages: + - python + max_files: 2000 + - owner: facebookresearch + name: dlrm + languages: + - python + max_files: 1000 + - owner: facebookresearch + name: ReAgent + languages: + - python + max_files: 1500 + python_tools: + description: Python tooling (black, mypy, ruff, pytest) + repos: + - owner: psf + name: black + languages: + - python + max_files: 2000 + - owner: psf + name: requests + languages: + - python + max_files: 1500 + - owner: psf + name: urllib3 + languages: + - python + max_files: 1500 + - owner: pycqa + name: pylint + languages: + - python + max_files: 3000 + - owner: pycqa + name: flake8 + languages: + - python + max_files: 1500 + - owner: pycqa + name: isort + languages: + - python + max_files: 500 + - owner: pycqa + name: bandit + languages: + - python + max_files: 800 + - owner: pycqa + name: coveragepy + languages: + - python + max_files: 2000 + - owner: astral-sh + name: ruff + languages: + - rust + max_files: 1500 + - owner: astral-sh + name: uv + languages: + - rust + max_files: 1500 + - owner: pre-commit + name: pre-commit + languages: + - python + max_files: 1000 + - owner: pytest + name: pytest + languages: + - python + max_files: 3000 + - owner: pytest-dev + name: pytest-cov + languages: + - python + max_files: 500 + - owner: pytest-dev + name: pytest-asyncio + languages: + - python + max_files: 500 + - owner: pytest-dev + name: pytest-xdist + languages: + - python + max_files: 800 + - owner: pytest-dev + name: pytest-mock + languages: + - python + max_files: 500 + - owner: pytest-dev + name: pytest-flask + languages: + - python + max_files: 400 + - owner: pytest-dev + name: pytest-django + languages: + - python + max_files: 800 + - owner: tox-dev + name: tox + languages: + - python + max_files: 1500 + - owner: python-poetry + name: poetry + languages: + - python + max_files: 4000 + - owner: pipx + name: pipx + languages: + - python + max_files: 800 + - owner: pypa + name: twine + languages: + - python + max_files: 800 + - owner: pypa + name: warehouse + languages: + - python + max_files: 5000 + - owner: conda + name: conda + languages: + - python + max_files: 5000 + - owner: conda + name: conda-build + languages: + - python + max_files: 2000 + - owner: pyenv + name: pyenv + languages: + - shell + max_files: 800 + - owner: asdf-vm + name: asdf + languages: + - shell + max_files: 500 + - owner: spulec + name: moto + languages: + - python + max_files: 3000 + - owner: getsentry + name: sentry-python + languages: + - python + max_files: 1500 + - owner: open-telemetry + name: opentelemetry-python + languages: + - python + max_files: 2500 + javascript_core: + description: JS/TS core (node, deno, v8, typescript) + repos: + - owner: nodejs + name: node + languages: + - javascript + - cpp + max_files: 8000 + - owner: denoland + name: deno + languages: + - rust + - typescript + max_files: 5000 + - owner: oven-sh + name: bun + languages: + - zig + - typescript + max_files: 3000 + - owner: microsoft + name: TypeScript + languages: + - typescript + max_files: 10000 + - owner: babel + name: babel + languages: + - javascript + max_files: 6000 + - owner: eslint + name: eslint + languages: + - javascript + max_files: 5000 + - owner: eslint + name: espree + languages: + - javascript + max_files: 1000 + - owner: jquery + name: jquery + languages: + - javascript + max_files: 2000 + - owner: lodash + name: lodash + languages: + - javascript + max_files: 2000 + - owner: documentcloud + name: underscore + languages: + - javascript + max_files: 1500 + - owner: moment + name: moment + languages: + - javascript + max_files: 2000 + - owner: date-fns + name: date-fns + languages: + - typescript + max_files: 2000 + - owner: axios + name: axios + languages: + - javascript + max_files: 1500 + - owner: ReactiveX + name: rxjs + languages: + - typescript + max_files: 3000 + - owner: immutable-js + name: immutable-js + languages: + - typescript + max_files: 2000 + - owner: immerjs + name: immer + languages: + - typescript + max_files: 1000 + - owner: prettier + name: prettier + languages: + - typescript + max_files: 4000 + - owner: webpack + name: webpack + languages: + - javascript + - typescript + max_files: 5000 + - owner: webpack + name: webpack-cli + languages: + - javascript + max_files: 1500 + - owner: webpack + name: webpack-dev-server + languages: + - javascript + max_files: 1500 + - owner: evanw + name: esbuild + languages: + - go + - javascript + max_files: 1500 + - owner: rollup + name: rollup + languages: + - javascript + - typescript + max_files: 2500 + - owner: vitejs + name: vite + languages: + - typescript + max_files: 3000 + - owner: parcel-bundler + name: parcel + languages: + - javascript + - rust + max_files: 4000 + - owner: swc-project + name: swc + languages: + - rust + max_files: 3000 + - owner: denoland + name: deno_std + languages: + - typescript + max_files: 3000 + - owner: denoland + name: deno_lint + languages: + - rust + max_files: 1000 + - owner: microsoft + name: ts-node + languages: + - typescript + max_files: 1000 + - owner: typestack + name: class-transformer + languages: + - typescript + max_files: 800 + - owner: typestack + name: class-validator + languages: + - typescript + max_files: 800 + - owner: nestjs + name: nest + languages: + - typescript + max_files: 5000 + - owner: nestjs + name: nx + languages: + - typescript + max_files: 4000 + - owner: trpc + name: trpc + languages: + - typescript + max_files: 2500 + - owner: prisma + name: prisma + languages: + - typescript + - rust + max_files: 5000 + javascript_frameworks: + description: JS frameworks (React, Vue, Angular, ...) + repos: + - owner: facebook + name: react + languages: + - javascript + - typescript + max_files: 8000 + - owner: facebook + name: react-native + languages: + - javascript + max_files: 8000 + - owner: facebook + name: relay + languages: + - javascript + max_files: 3000 + - owner: facebook + name: flux + languages: + - javascript + max_files: 1000 + - owner: facebook + name: jest + languages: + - javascript + - typescript + max_files: 5000 + - owner: facebook + name: docusaurus + languages: + - javascript + - typescript + max_files: 3000 + - owner: facebook + name: create-react-app + languages: + - javascript + max_files: 2000 + - owner: facebook + name: react-devtools + languages: + - javascript + max_files: 1500 + - owner: vuejs + name: vue + languages: + - javascript + - typescript + max_files: 5000 + - owner: vuejs + name: vue-next + languages: + - typescript + max_files: 4000 + - owner: vuejs + name: vite + languages: + - typescript + max_files: 3000 + - owner: vuejs + name: vue-router + languages: + - typescript + max_files: 1500 + - owner: vuejs + name: vuex + languages: + - javascript + max_files: 1500 + - owner: vuejs + name: pinia + languages: + - typescript + max_files: 1000 + - owner: vuejs + name: vue-cli + languages: + - javascript + max_files: 3000 + - owner: vuejs + name: vuepress + languages: + - javascript + max_files: 2000 + - owner: vuejs + name: vue-test-utils + languages: + - typescript + max_files: 800 + - owner: vuejs + name: volar + languages: + - typescript + max_files: 1500 + - owner: angular + name: angular + languages: + - typescript + max_files: 10000 + - owner: angular + name: angular-cli + languages: + - typescript + max_files: 3000 + - owner: angular + name: angular.js + languages: + - javascript + max_files: 5000 + - owner: angular + name: material + languages: + - typescript + max_files: 4000 + - owner: angular + name: universal + languages: + - typescript + max_files: 1000 + - owner: sveltejs + name: svelte + languages: + - javascript + max_files: 3000 + - owner: sveltejs + name: kit + languages: + - javascript + max_files: 2000 + - owner: sveltejs + name: language-tools + languages: + - typescript + max_files: 800 + - owner: solidjs + name: solid + languages: + - typescript + max_files: 1500 + - owner: solidjs + name: solid-start + languages: + - typescript + max_files: 1000 + - owner: preactjs + name: preact + languages: + - javascript + max_files: 2000 + - owner: emberjs + name: ember.js + languages: + - javascript + max_files: 4000 + - owner: emberjs + name: data + languages: + - javascript + max_files: 2000 + - owner: emberjs + name: ember-cli + languages: + - javascript + max_files: 2000 + - owner: backbonejs + name: backbone + languages: + - javascript + max_files: 1500 + - owner: jashkenas + name: backbone + languages: + - javascript + max_files: 1500 + - owner: aurelia + name: framework + languages: + - typescript + max_files: 2000 + - owner: meteor + name: meteor + languages: + - javascript + max_files: 5000 + - owner: polymer + name: polymer + languages: + - javascript + max_files: 3000 + - owner: PolymerLabs + name: lit-html + languages: + - javascript + max_files: 1500 + - owner: lit + name: lit + languages: + - typescript + max_files: 1500 + - owner: vercel + name: next.js + languages: + - javascript + - typescript + max_files: 5000 + - owner: vercel + name: swr + languages: + - typescript + max_files: 1000 + - owner: vercel + name: ai + languages: + - typescript + max_files: 2000 + - owner: vercel + name: turborepo + languages: + - rust + - typescript + max_files: 2500 + - owner: nuxt + name: nuxt.js + languages: + - javascript + - typescript + max_files: 4000 + - owner: nuxt + name: framework + languages: + - typescript + max_files: 2500 + - owner: remix-run + name: remix + languages: + - typescript + max_files: 3000 + - owner: remix-run + name: react-router + languages: + - typescript + max_files: 2000 + - owner: gatsbyjs + name: gatsby + languages: + - javascript + - typescript + max_files: 5000 + - owner: withastro + name: astro + languages: + - typescript + max_files: 3000 + - owner: withastro + name: compiler + languages: + - rust + max_files: 800 + - owner: 11ty + name: eleventy + languages: + - javascript + max_files: 1500 + - owner: storybookjs + name: storybook + languages: + - javascript + - typescript + max_files: 8000 + - owner: mui-org + name: material-ui + languages: + - javascript + - typescript + max_files: 5000 + - owner: ant-design + name: ant-design + languages: + - typescript + max_files: 5000 + - owner: ant-design + name: ant-design-pro + languages: + - typescript + max_files: 2500 + - owner: chakra-ui + name: chakra-ui + languages: + - typescript + max_files: 3000 + - owner: tailwindlabs + name: tailwindcss + languages: + - javascript + - typescript + max_files: 3000 + - owner: tailwindlabs + name: headlessui + languages: + - javascript + - typescript + max_files: 1500 + - owner: tailwindlabs + name: heroicons + languages: + - javascript + max_files: 500 + - owner: twbs + name: bootstrap + languages: + - javascript + - scss + max_files: 5000 + - owner: chartjs + name: Chart.js + languages: + - javascript + max_files: 3000 + - owner: apexcharts + name: apexcharts.js + languages: + - javascript + max_files: 3000 + - owner: d3 + name: d3 + languages: + - javascript + max_files: 5000 + - owner: d3 + name: d3-array + languages: + - javascript + max_files: 800 + - owner: d3 + name: d3-scale + languages: + - javascript + max_files: 500 + - owner: d3 + name: d3-shape + languages: + - javascript + max_files: 800 + - owner: d3 + name: d3-format + languages: + - javascript + max_files: 300 + - owner: d3 + name: d3-time + languages: + - javascript + max_files: 500 + - owner: d3 + name: d3-color + languages: + - javascript + max_files: 500 + - owner: d3 + name: d3-interpolate + languages: + - javascript + max_files: 500 + - owner: d3 + name: d3-selection + languages: + - javascript + max_files: 500 + - owner: d3 + name: d3-transition + languages: + - javascript + max_files: 500 + - owner: d3 + name: d3-axis + languages: + - javascript + max_files: 300 + - owner: d3 + name: d3-drag + languages: + - javascript + max_files: 300 + - owner: d3 + name: d3-zoom + languages: + - javascript + max_files: 500 + - owner: d3 + name: d3-fetch + languages: + - javascript + max_files: 200 + - owner: d3 + name: d3-dsv + languages: + - javascript + max_files: 300 + - owner: d3 + name: d3-random + languages: + - javascript + max_files: 300 + - owner: d3 + name: d3-quadtree + languages: + - javascript + max_files: 300 + - owner: d3 + name: d3-hierarchy + languages: + - javascript + max_files: 500 + - owner: d3 + name: d3-geo + languages: + - javascript + max_files: 800 + - owner: d3 + name: d3-voronoi + languages: + - javascript + max_files: 300 + - owner: d3 + name: d3-force + languages: + - javascript + max_files: 500 + - owner: d3 + name: d3-delaunay + languages: + - javascript + max_files: 500 + - owner: recharts + name: recharts + languages: + - javascript + max_files: 2500 + - owner: plotly + name: plotly.js + languages: + - javascript + max_files: 5000 + - owner: vega + name: vega + languages: + - javascript + max_files: 2500 + - owner: vega + name: vega-lite + languages: + - javascript + - typescript + max_files: 3000 + - owner: vega + name: vega-embed + languages: + - javascript + max_files: 500 + rust_core: + description: Rust core, std, ecosystem + repos: + - owner: rust-lang + name: rust + languages: + - rust + max_files: 20000 + - owner: rust-lang + name: cargo + languages: + - rust + max_files: 4000 + - owner: rust-lang + name: book + languages: + - markdown + - rust + max_files: 1500 + - owner: rust-lang + name: rustlings + languages: + - rust + max_files: 800 + - owner: rust-lang + name: rustup + languages: + - rust + max_files: 2000 + - owner: rust-lang + name: rustfmt + languages: + - rust + max_files: 1500 + - owner: rust-lang + name: rust-clippy + languages: + - rust + max_files: 2500 + - owner: rust-lang + name: miri + languages: + - rust + max_files: 1500 + - owner: rust-lang + name: rust-analyzer + languages: + - rust + max_files: 5000 + - owner: rust-lang + name: chalk + languages: + - rust + max_files: 1500 + - owner: rust-lang + name: rfcs + languages: + - markdown + max_files: 1000 + - owner: rust-lang + name: crates.io + languages: + - rust + max_files: 4000 + - owner: rust-lang-nursery + name: rand + languages: + - rust + max_files: 1500 + - owner: rust-random + name: rand + languages: + - rust + max_files: 1500 + - owner: rust-itertools + name: itertools + languages: + - rust + max_files: 1000 + - owner: serde-rs + name: serde + languages: + - rust + max_files: 2000 + - owner: serde-rs + name: serde-json + languages: + - rust + max_files: 1500 + - owner: serde-rs + name: serde-yaml + languages: + - rust + max_files: 500 + - owner: tokio-rs + name: tokio + languages: + - rust + max_files: 4000 + - owner: tokio-rs + name: tokio-util + languages: + - rust + max_files: 800 + - owner: tokio-rs + name: bytes + languages: + - rust + max_files: 800 + - owner: tokio-rs + name: mio + languages: + - rust + max_files: 1500 + - owner: tokio-rs + name: tracing + languages: + - rust + max_files: 2000 + - owner: tokio-rs + name: tracing-subscriber + languages: + - rust + max_files: 800 + - owner: tokio-rs + name: axum + languages: + - rust + max_files: 1500 + - owner: tokio-rs + name: tower + languages: + - rust + max_files: 1500 + - owner: tokio-rs + name: tower-http + languages: + - rust + max_files: 800 + - owner: hyperium + name: hyper + languages: + - rust + max_files: 3000 + - owner: hyperium + name: tonic + languages: + - rust + max_files: 2000 + - owner: hyperium + name: h2 + languages: + - rust + max_files: 1500 + - owner: hyperium + name: http + languages: + - rust + max_files: 800 + - owner: actix + name: actix-web + languages: + - rust + max_files: 3000 + - owner: actix + name: actix + languages: + - rust + max_files: 1500 + - owner: actix + name: actix-http + languages: + - rust + max_files: 1000 + - owner: actix + name: actix-rt + languages: + - rust + max_files: 400 + - owner: BurntSushi + name: ripgrep + languages: + - rust + max_files: 1500 + - owner: BurntSushi + name: toml-rs + languages: + - rust + max_files: 800 + - owner: BurntSushi + name: csv + languages: + - rust + max_files: 1000 + - owner: BurntSushi + name: regex + languages: + - rust + max_files: 1500 + - owner: clap-rs + name: clap + languages: + - rust + max_files: 2000 + - owner: rayon-rs + name: rayon + languages: + - rust + max_files: 1500 + - owner: crossbeam-rs + name: crossbeam + languages: + - rust + max_files: 2000 + - owner: crossbeam-rs + name: crossbeam-channel + languages: + - rust + max_files: 800 + - owner: rust-windowing + name: winit + languages: + - rust + max_files: 2500 + - owner: rust-windowing + name: glutin + languages: + - rust + max_files: 1500 + - owner: gfx-rs + name: wgpu + languages: + - rust + max_files: 4000 + - owner: bevyengine + name: bevy + languages: + - rust + max_files: 8000 + - owner: amethyst + name: amethyst + languages: + - rust + max_files: 4000 + - owner: ggez + name: ggez + languages: + - rust + max_files: 1500 + - owner: image-rs + name: image + languages: + - rust + max_files: 2500 + - owner: image-rs + name: imageproc + languages: + - rust + max_files: 1500 + - owner: rust-ml + name: linfa + languages: + - rust + max_files: 2000 + - owner: rust-ml + name: smartcore + languages: + - rust + max_files: 1500 + - owner: tract + name: tract + languages: + - rust + max_files: 2500 + - owner: dtolnay + name: anyhow + languages: + - rust + max_files: 500 + - owner: dtolnay + name: thiserror + languages: + - rust + max_files: 500 + - owner: dtolnay + name: syn + languages: + - rust + max_files: 2500 + - owner: dtolnay + name: quote + languages: + - rust + max_files: 400 + - owner: dtolnay + name: proc-macro2 + languages: + - rust + max_files: 500 + go_core: + description: Go core, stdlib, ecosystem + repos: + - owner: golang + name: go + languages: + - go + - c + max_files: 15000 + - owner: golang + name: tools + languages: + - go + max_files: 3000 + - owner: golang + name: mock + languages: + - go + max_files: 800 + - owner: golang + name: lint + languages: + - go + max_files: 800 + - owner: golang + name: proposal + languages: + - markdown + max_files: 800 + - owner: golang + name: vuln + languages: + - go + max_files: 500 + - owner: golang + name: example + languages: + - go + max_files: 1500 + - owner: golang + name: playground + languages: + - go + max_files: 500 + - owner: golang + name: oauth2 + languages: + - go + max_files: 800 + - owner: golang + name: net + languages: + - go + max_files: 2000 + - owner: golang + name: crypto + languages: + - go + max_files: 1500 + - owner: golang + name: sys + languages: + - go + max_files: 1500 + - owner: golang + name: text + languages: + - go + max_files: 1500 + - owner: golang + name: image + languages: + - go + max_files: 1000 + - owner: golang + name: exp + languages: + - go + max_files: 1500 + - owner: golang + name: mobile + languages: + - go + max_files: 800 + - owner: golang + name: protobuf + languages: + - go + max_files: 1500 + - owner: golang + name: grpc-go + languages: + - go + max_files: 2500 + - owner: gin-gonic + name: gin + languages: + - go + max_files: 2000 + - owner: labstack + name: echo + languages: + - go + max_files: 2000 + - owner: gofiber + name: fiber + languages: + - go + max_files: 2000 + - owner: go-chi + name: chi + languages: + - go + max_files: 1000 + - owner: gorilla + name: mux + languages: + - go + max_files: 800 + - owner: gorilla + name: websocket + languages: + - go + max_files: 800 + - owner: gorilla + name: handlers + languages: + - go + max_files: 300 + - owner: gorilla + name: context + languages: + - go + max_files: 200 + - owner: gorilla + name: sessions + languages: + - go + max_files: 500 + - owner: gorilla + name: securecookie + languages: + - go + max_files: 200 + - owner: gorilla + name: csrf + languages: + - go + max_files: 300 + - owner: emicklei + name: go-restful + languages: + - go + max_files: 1500 + - owner: go-kit + name: kit + languages: + - go + max_files: 2000 + - owner: micro + name: go-micro + languages: + - go + max_files: 3000 + - owner: micro + name: micro + languages: + - go + max_files: 4000 + - owner: go-kratos + name: kratos + languages: + - go + max_files: 3000 + - owner: zeromicro + name: go-zero + languages: + - go + max_files: 2500 + - owner: flamego + name: flamego + languages: + - go + max_files: 1000 + - owner: gobuffalo + name: buffalo + languages: + - go + max_files: 2500 + - owner: gobuffalo + name: pop + languages: + - go + max_files: 1500 + - owner: spf13 + name: cobra + languages: + - go + max_files: 2000 + - owner: spf13 + name: viper + languages: + - go + max_files: 2000 + - owner: spf13 + name: cast + languages: + - go + max_files: 300 + - owner: spf13 + name: afero + languages: + - go + max_files: 800 + - owner: spf13 + name: pflag + languages: + - go + max_files: 500 + - owner: urfave + name: cli + languages: + - go + max_files: 2000 + - owner: mitchellh + name: mapstructure + languages: + - go + max_files: 800 + - owner: hashicorp + name: consul + languages: + - go + max_files: 5000 + - owner: hashicorp + name: vault + languages: + - go + max_files: 6000 + - owner: hashicorp + name: terraform + languages: + - go + max_files: 8000 + - owner: hashicorp + name: nomad + languages: + - go + max_files: 5000 + - owner: hashicorp + name: packer + languages: + - go + max_files: 4000 + - owner: hashicorp + name: hcl + languages: + - go + max_files: 1500 + - owner: hashicorp + name: go-plugin + languages: + - go + max_files: 1000 + - owner: hashicorp + name: go-getter + languages: + - go + max_files: 1000 + - owner: hashicorp + name: raft + languages: + - go + max_files: 2000 + - owner: hashicorp + name: memberlist + languages: + - go + max_files: 1000 + - owner: hashicorp + name: serf + languages: + - go + max_files: 1500 + - owner: docker + name: docker + languages: + - go + max_files: 10000 + - owner: docker + name: compose + languages: + - go + max_files: 5000 + - owner: docker + name: distribution + languages: + - go + max_files: 3000 + - owner: docker + name: cli + languages: + - go + max_files: 3000 + - owner: docker + name: buildx + languages: + - go + max_files: 1500 + - owner: docker + name: swarmkit + languages: + - go + max_files: 3000 + - owner: moby + name: moby + languages: + - go + max_files: 12000 + - owner: moby + name: buildkit + languages: + - go + max_files: 4000 + - owner: containerd + name: containerd + languages: + - go + max_files: 6000 + - owner: containerd + name: nerdctl + languages: + - go + max_files: 1500 + - owner: k3s-io + name: k3s + languages: + - go + max_files: 3000 + - owner: rancher + name: rke2 + languages: + - go + max_files: 1500 + - owner: rancher + name: rancher + languages: + - go + max_files: 5000 + - owner: rancher + name: fleet + languages: + - go + max_files: 2000 + - owner: opencontainers + name: runc + languages: + - go + max_files: 3000 + - owner: opencontainers + name: image-spec + languages: + - go + - markdown + max_files: 800 + - owner: opencontainers + name: runtime-spec + languages: + - go + - markdown + max_files: 800 + - owner: opencontainers + name: selinux + languages: + - go + max_files: 200 + - owner: opencontainers + name: go-digest + languages: + - go + max_files: 200 + - owner: opencontainers + name: storage + languages: + - go + max_files: 1500 + java_core: + description: Java core, Spring, Hibernate, Quarkus + repos: + - owner: openjdk + name: jdk + languages: + - java + - c + max_files: 15000 + - owner: openjdk + name: jol + languages: + - java + max_files: 800 + - owner: openjdk + name: skara + languages: + - java + max_files: 800 + - owner: eclipse-ee4j + name: jpa-api + languages: + - java + max_files: 500 + - owner: eclipse-ee4j + name: jaxrs-api + languages: + - java + max_files: 500 + - owner: eclipse-ee4j + name: servlet-api + languages: + - java + max_files: 500 + - owner: spring-projects + name: spring-framework + languages: + - java + max_files: 8000 + - owner: spring-projects + name: spring-boot + languages: + - java + max_files: 8000 + - owner: spring-projects + name: spring-data-jpa + languages: + - java + max_files: 2000 + - owner: spring-projects + name: spring-data-mongodb + languages: + - java + max_files: 1500 + - owner: spring-projects + name: spring-data-redis + languages: + - java + max_files: 1500 + - owner: spring-projects + name: spring-data-elasticsearch + languages: + - java + max_files: 1500 + - owner: spring-projects + name: spring-security + languages: + - java + max_files: 5000 + - owner: spring-projects + name: spring-cloud + languages: + - java + max_files: 3000 + - owner: spring-projects + name: spring-graphql + languages: + - java + max_files: 1500 + - owner: spring-projects + name: spring-kafka + languages: + - java + max_files: 1500 + - owner: spring-projects + name: spring-amqp + languages: + - java + max_files: 1000 + - owner: spring-projects + name: spring-session + languages: + - java + max_files: 1500 + - owner: spring-projects + name: spring-batch + languages: + - java + max_files: 2500 + - owner: spring-projects + name: spring-integration + languages: + - java + max_files: 3000 + - owner: spring-projects + name: spring-shell + languages: + - java + max_files: 800 + - owner: spring-projects + name: spring-hateoas + languages: + - java + max_files: 800 + - owner: spring-projects + name: spring-vault + languages: + - java + max_files: 800 + - owner: spring-projects + name: spring-statemachine + languages: + - java + max_files: 1500 + - owner: hibernate + name: hibernate-orm + languages: + - java + max_files: 5000 + - owner: hibernate + name: hibernate-validator + languages: + - java + max_files: 2000 + - owner: hibernate + name: hibernate-search + languages: + - java + max_files: 2500 + - owner: hibernate + name: hibernate-reactive + languages: + - java + max_files: 1500 + - owner: quarkusio + name: quarkus + languages: + - java + max_files: 8000 + - owner: quarkusio + name: quarkus-quickstarts + languages: + - java + max_files: 1500 + - owner: eclipse-vertx + name: vert.x + languages: + - java + max_files: 5000 + - owner: micronaut-projects + name: micronaut-core + languages: + - java + - groovy + max_files: 5000 + - owner: micronaut-projects + name: micronaut-spring + languages: + - java + max_files: 800 + - owner: micronaut-projects + name: micronaut-data + languages: + - java + max_files: 1500 + - owner: micronaut-projects + name: micronaut-security + languages: + - java + max_files: 1500 + - owner: apache + name: kafka + languages: + - java + - scala + max_files: 8000 + - owner: apache + name: camel + languages: + - java + max_files: 10000 + - owner: apache + name: spark + languages: + - scala + - java + max_files: 15000 + - owner: apache + name: flink + languages: + - java + max_files: 10000 + - owner: apache + name: beam + languages: + - java + - python + max_files: 10000 + - owner: apache + name: hadoop + languages: + - java + - shell + max_files: 8000 + - owner: apache + name: hbase + languages: + - java + max_files: 5000 + - owner: apache + name: cassandra + languages: + - java + max_files: 8000 + - owner: apache + name: tomcat + languages: + - java + max_files: 5000 + - owner: apache + name: maven + languages: + - java + max_files: 5000 + - owner: apache + name: groovy + languages: + - java + - groovy + max_files: 5000 + - owner: apache + name: jmeter + languages: + - java + max_files: 5000 + - owner: apache + name: lucene + languages: + - java + max_files: 5000 + - owner: apache + name: solr + languages: + - java + max_files: 5000 + - owner: apache + name: commons-lang + languages: + - java + max_files: 1500 + - owner: apache + name: commons-io + languages: + - java + max_files: 1000 + - owner: apache + name: commons-collections + languages: + - java + max_files: 1500 + - owner: apache + name: commons-codec + languages: + - java + max_files: 500 + - owner: apache + name: commons-compress + languages: + - java + max_files: 1500 + - owner: apache + name: commons-math + languages: + - java + max_files: 2500 + - owner: apache + name: commons-net + languages: + - java + max_files: 800 + - owner: apache + name: commons-text + languages: + - java + max_files: 800 + - owner: apache + name: commons-csv + languages: + - java + max_files: 500 + - owner: apache + name: commons-dbutils + languages: + - java + max_files: 300 + - owner: apache + name: commons-fileupload + languages: + - java + max_files: 500 + - owner: apache + name: commons-beanutils + languages: + - java + max_files: 800 + - owner: apache + name: commons-validator + languages: + - java + max_files: 800 + - owner: apache + name: commons-exec + languages: + - java + max_files: 400 + - owner: apache + name: commons-configuration + languages: + - java + max_files: 1000 + - owner: apache + name: httpcomponents-client + languages: + - java + max_files: 2000 + - owner: apache + name: httpcomponents-core + languages: + - java + max_files: 1500 + - owner: google + name: guava + languages: + - java + max_files: 5000 + - owner: google + name: gson + languages: + - java + max_files: 1500 + - owner: google + name: protobuf + languages: + - java + - c++ + - python + max_files: 5000 + - owner: google + name: dagger + languages: + - java + max_files: 1500 + - owner: google + name: auto + languages: + - java + max_files: 1500 + - owner: google + name: tink + languages: + - java + - go + - python + max_files: 2500 + - owner: google + name: closure-compiler + languages: + - java + - javascript + max_files: 5000 + - owner: google + name: error-prone + languages: + - java + max_files: 2000 + - owner: google + name: flogger + languages: + - java + max_files: 1000 + - owner: google + name: guice + languages: + - java + max_files: 3000 + - owner: google + name: google-java-format + languages: + - java + max_files: 1000 + - owner: JetBrains + name: kotlin + languages: + - kotlin + - java + max_files: 15000 + - owner: JetBrains + name: Exposed + languages: + - kotlin + max_files: 1500 + - owner: JetBrains + name: compose-jb + languages: + - kotlin + max_files: 3000 + - owner: JetBrains + name: intellij-community + languages: + - java + - kotlin + max_files: 15000 + - owner: Kotlin + name: kotlinx.coroutines + languages: + - kotlin + max_files: 2000 + - owner: Kotlin + name: kotlinx.serialization + languages: + - kotlin + max_files: 2000 + - owner: Kotlin + name: kotlinx.html + languages: + - kotlin + max_files: 800 + - owner: Kotlin + name: kotlinx.atomicfu + languages: + - kotlin + max_files: 500 + - owner: Kotlin + name: kotlinx-datetime + languages: + - kotlin + max_files: 500 + - owner: Kotlin + name: kotlinx-io + languages: + - kotlin + max_files: 800 + c_cpp: + description: C / C++ (gcc, llvm, boost, postgres) + repos: + - owner: gcc + name: gcc + languages: + - c + - c++ + max_files: 15000 + - owner: gcc + name: glibc + languages: + - c + max_files: 8000 + - owner: llvm + name: llvm-project + languages: + - c++ + max_files: 30000 + - owner: llvm + name: compiler-rt + languages: + - c++ + max_files: 2000 + - owner: llvm + name: libcxx + languages: + - c++ + max_files: 2500 + - owner: llvm + name: libcxxabi + languages: + - c++ + max_files: 800 + - owner: llvm + name: libunwind + languages: + - c++ + max_files: 800 + - owner: llvm + name: openmp + languages: + - c++ + max_files: 1500 + - owner: llvm + name: lld + languages: + - c++ + max_files: 1500 + - owner: llvm + name: lldb + languages: + - c++ + max_files: 3000 + - owner: llvm + name: polly + languages: + - c++ + max_files: 1000 + - owner: llvm + name: mlir + languages: + - c++ + max_files: 3000 + - owner: llvm + name: flang + languages: + - c++ + - fortran + max_files: 1500 + - owner: llvm + name: bolt + languages: + - c++ + max_files: 1000 + - owner: boostorg + name: boost + languages: + - c++ + max_files: 15000 + - owner: boostorg + name: beast + languages: + - c++ + max_files: 1500 + - owner: boostorg + name: asio + languages: + - c++ + max_files: 1500 + - owner: boostorg + name: system + languages: + - c++ + max_files: 300 + - owner: boostorg + name: core + languages: + - c++ + max_files: 500 + - owner: boostorg + name: config + languages: + - c++ + max_files: 800 + - owner: boostorg + name: preprocessor + languages: + - c++ + max_files: 800 + - owner: boostorg + name: mpl + languages: + - c++ + max_files: 1000 + - owner: boostorg + name: fusion + languages: + - c++ + max_files: 1500 + - owner: boostorg + name: hana + languages: + - c++ + max_files: 1000 + - owner: boostorg + name: geometry + languages: + - c++ + max_files: 3000 + - owner: boostorg + name: multiprecision + languages: + - c++ + max_files: 1500 + - owner: boostorg + name: math + languages: + - c++ + max_files: 3000 + - owner: boostorg + name: random + languages: + - c++ + max_files: 500 + - owner: boostorg + name: filesystem + languages: + - c++ + max_files: 500 + - owner: boostorg + name: thread + languages: + - c++ + max_files: 1500 + - owner: boostorg + name: log + languages: + - c++ + max_files: 2000 + - owner: boostorg + name: program_options + languages: + - c++ + max_files: 800 + - owner: boostorg + name: test + languages: + - c++ + max_files: 2000 + - owner: boostorg + name: regex + languages: + - c++ + max_files: 1000 + - owner: boostorg + name: property_tree + languages: + - c++ + max_files: 500 + - owner: boostorg + name: json + languages: + - c++ + max_files: 800 + - owner: boostorg + name: url + languages: + - c++ + max_files: 500 + - owner: boostorg + name: nowide + languages: + - c++ + max_files: 300 + - owner: boostorg + name: leaf + languages: + - c++ + max_files: 300 + - owner: boostorg + name: pfr + languages: + - c++ + max_files: 500 + - owner: boostorg + name: spirit + languages: + - c++ + max_files: 3000 + - owner: boostorg + name: proto + languages: + - c++ + max_files: 1500 + - owner: catchorg + name: Catch2 + languages: + - c++ + max_files: 1500 + - owner: catchorg + name: Clara + languages: + - c++ + max_files: 300 + - owner: google + name: googletest + languages: + - c++ + max_files: 3000 + - owner: google + name: googlemock + languages: + - c++ + max_files: 1000 + - owner: google + name: benchmark + languages: + - c++ + max_files: 1500 + - owner: google + name: abseil-cpp + languages: + - c++ + max_files: 3000 + - owner: google + name: boringssl + languages: + - c + - c++ + max_files: 2500 + - owner: facebook + name: folly + languages: + - c++ + max_files: 5000 + - owner: facebook + name: rocksdb + languages: + - c++ + max_files: 4000 + - owner: facebook + name: fbthrift + languages: + - c++ + max_files: 2000 + - owner: facebook + name: proxygen + languages: + - c++ + max_files: 2000 + - owner: facebook + name: wangle + languages: + - c++ + max_files: 1500 + - owner: facebook + name: mcrouter + languages: + - c++ + max_files: 1500 + - owner: facebook + name: mvfst + languages: + - c++ + max_files: 2000 + - owner: facebook + name: hhvm + languages: + - c++ + - php + max_files: 5000 + - owner: facebook + name: react-native + languages: + - c++ + - javascript + max_files: 6000 + - owner: facebook + name: yoga + languages: + - c++ + - javascript + max_files: 1500 + - owner: facebook + name: flipper + languages: + - c++ + - javascript + max_files: 3000 + - owner: facebook + name: hermes + languages: + - c++ + max_files: 3000 + - owner: facebookincubator + name: fbzmq + languages: + - c++ + max_files: 500 + - owner: facebookincubator + name: oomd + languages: + - c++ + max_files: 800 + - owner: facebookincubator + name: katran + languages: + - c++ + max_files: 1000 + - owner: facebookresearch + name: faiss + languages: + - c++ + max_files: 4000 + - owner: facebookresearch + name: flashlight + languages: + - c++ + max_files: 3000 + - owner: NVIDIA + name: cutlass + languages: + - c++ + max_files: 2500 + - owner: NVIDIA + name: cub + languages: + - c++ + max_files: 1500 + - owner: NVIDIA + name: thrust + languages: + - c++ + max_files: 2500 + - owner: NVIDIA + name: nccl + languages: + - c++ + - cuda + max_files: 1500 + - owner: NVIDIA + name: nvbench + languages: + - c++ + max_files: 500 + - owner: NVIDIA + name: FasterTransformer + languages: + - c++ + - cuda + max_files: 2500 + - owner: NVIDIA + name: apex + languages: + - python + - c++ + - cuda + max_files: 1500 + - owner: NVIDIA + name: DeepLearningExamples + languages: + - python + max_files: 3000 + - owner: NVIDIA + name: DALI + languages: + - c++ + - python + max_files: 3000 + - owner: NVIDIA + name: TensorRT-LLM + languages: + - c++ + - python + max_files: 2500 + - owner: NVIDIA + name: cuda-samples + languages: + - c++ + - cuda + max_files: 1500 + - owner: torvalds + name: linux + languages: + - c + max_files: 20000 + - owner: git + name: git + languages: + - c + - shell + max_files: 5000 + - owner: redis + name: redis + languages: + - c + max_files: 4000 + - owner: redis + name: hiredis + languages: + - c + max_files: 800 + - owner: redis + name: jedis + languages: + - java + max_files: 2000 + - owner: redis + name: lettuce + languages: + - java + max_files: 2500 + - owner: antirez + name: kilo + languages: + - c + max_files: 500 + - owner: antirez + name: sd + languages: + - c + max_files: 800 + - owner: antirez + name: disque + languages: + - c + max_files: 800 + - owner: sqlite + name: sqlite + languages: + - c + max_files: 5000 + - owner: sqlite + name: fossil + languages: + - c + max_files: 3000 + - owner: PostgreSQL + name: postgresql + languages: + - c + max_files: 10000 + - owner: postgres + name: pgvector + languages: + - c + max_files: 800 + - owner: mysql + name: mysql-server + languages: + - c + - c++ + max_files: 10000 + - owner: mongodb + name: mongo + languages: + - c++ + max_files: 8000 + - owner: mongodb + name: mongo-c-driver + languages: + - c + max_files: 2500 + - owner: mongodb + name: mongo-cxx-driver + languages: + - c++ + max_files: 2500 + - owner: elastic + name: elasticsearch + languages: + - java + max_files: 8000 + - owner: elastic + name: kibana + languages: + - typescript + max_files: 8000 + - owner: elastic + name: logstash + languages: + - java + - ruby + max_files: 5000 + - owner: elastic + name: beats + languages: + - go + max_files: 5000 + devops: + description: DevOps (k8s, helm, prometheus, grafana) + repos: + - owner: kubernetes + name: kubernetes + languages: + - go + max_files: 20000 + - owner: kubernetes + name: client-go + languages: + - go + max_files: 3000 + - owner: kubernetes + name: api + languages: + - go + max_files: 2000 + - owner: kubernetes + name: dashboard + languages: + - typescript + max_files: 3000 + - owner: kubernetes + name: ingress-nginx + languages: + - go + max_files: 2500 + - owner: kubernetes + name: ingress-gce + languages: + - go + max_files: 1000 + - owner: kubernetes + name: kube-aggregator + languages: + - go + max_files: 800 + - owner: kubernetes + name: metrics + languages: + - go + max_files: 500 + - owner: kubernetes + name: kube-state-metrics + languages: + - go + max_files: 1500 + - owner: kubernetes + name: node-problem-detector + languages: + - go + max_files: 800 + - owner: kubernetes + name: autoscaler + languages: + - go + max_files: 3000 + - owner: kubernetes + name: minikube + languages: + - go + max_files: 4000 + - owner: kubernetes + name: kind + languages: + - go + max_files: 1500 + - owner: kubernetes + name: kops + languages: + - go + max_files: 4000 + - owner: kubernetes + name: cluster-autoscaler + languages: + - go + max_files: 2000 + - owner: kubernetes + name: kompose + languages: + - go + max_files: 1500 + - owner: kubernetes + name: kube-openapi + languages: + - go + max_files: 800 + - owner: kubernetes + name: examples + languages: + - go + max_files: 800 + - owner: kubernetes + name: website + languages: + - markdown + - html + max_files: 5000 + - owner: kubernetes + name: enhancements + languages: + - markdown + max_files: 1500 + - owner: kubernetes + name: community + languages: + - markdown + max_files: 3000 + - owner: kubernetes + name: test-infra + languages: + - go + max_files: 3000 + - owner: kubernetes-sigs + name: cluster-api + languages: + - go + max_files: 3000 + - owner: kubernetes-sigs + name: controller-runtime + languages: + - go + max_files: 2000 + - owner: kubernetes-sigs + name: kubebuilder + languages: + - go + max_files: 2000 + - owner: kubernetes-sigs + name: kustomize + languages: + - go + max_files: 2500 + - owner: kubernetes-sigs + name: kind + languages: + - go + max_files: 1500 + - owner: kubernetes-sigs + name: krew + languages: + - go + max_files: 800 + - owner: kubernetes-sigs + name: cluster-api-provider-aws + languages: + - go + max_files: 1500 + - owner: kubernetes-sigs + name: cluster-api-provider-azure + languages: + - go + max_files: 1500 + - owner: kubernetes-sigs + name: cluster-api-provider-vsphere + languages: + - go + max_files: 1000 + - owner: kubernetes-sigs + name: cluster-api-provider-gcp + languages: + - go + max_files: 800 + - owner: kubernetes-sigs + name: external-dns + languages: + - go + max_files: 1500 + - owner: kubernetes-sigs + name: aws-load-balancer-controller + languages: + - go + max_files: 1000 + - owner: kubernetes-sigs + name: azure-service-operator + languages: + - go + max_files: 1500 + - owner: kubernetes-sigs + name: secrets-store-csi-driver + languages: + - go + max_files: 800 + - owner: kubernetes-sigs + name: gateway-api + languages: + - go + max_files: 1000 + - owner: kubernetes-sigs + name: kueue + languages: + - go + max_files: 1500 + - owner: kubernetes-sigs + name: training-operator + languages: + - go + max_files: 1000 + - owner: kubernetes-sigs + name: karpenter + languages: + - go + max_files: 2500 + - owner: helm + name: helm + languages: + - go + max_files: 4000 + - owner: helm + name: charts + languages: + - yaml + max_files: 5000 + - owner: prometheus + name: prometheus + languages: + - go + max_files: 8000 + - owner: prometheus + name: alertmanager + languages: + - go + max_files: 2000 + - owner: prometheus + name: node_exporter + languages: + - go + max_files: 1500 + - owner: prometheus + name: blackbox_exporter + languages: + - go + max_files: 800 + - owner: prometheus + name: pushgateway + languages: + - go + max_files: 800 + - owner: prometheus + name: statsd_exporter + languages: + - go + max_files: 800 + - owner: prometheus + name: mysqld_exporter + languages: + - go + max_files: 800 + - owner: prometheus + name: postgres_exporter + languages: + - go + max_files: 800 + - owner: prometheus + name: redis_exporter + languages: + - go + max_files: 800 + - owner: prometheus + name: mongodb_exporter + languages: + - go + max_files: 800 + - owner: prometheus + name: consul_exporter + languages: + - go + max_files: 500 + - owner: prometheus + name: elasticsearch_exporter + languages: + - go + max_files: 500 + - owner: prometheus + name: kafka_exporter + languages: + - go + max_files: 500 + - owner: prometheus + name: snmp_exporter + languages: + - go + max_files: 800 + - owner: prometheus + name: ipmi_exporter + languages: + - go + max_files: 500 + - owner: prometheus + name: jmx_exporter + languages: + - java + max_files: 1500 + - owner: prometheus + name: client_golang + languages: + - go + max_files: 1500 + - owner: prometheus + name: client_python + languages: + - python + max_files: 800 + - owner: prometheus + name: client_java + languages: + - java + max_files: 1500 + - owner: prometheus + name: client_ruby + languages: + - ruby + max_files: 500 + - owner: prometheus + name: client_nodejs + languages: + - typescript + max_files: 500 + - owner: prometheus + name: client_php + languages: + - php + max_files: 300 + - owner: prometheus + name: client_csharp + languages: + - c# + max_files: 800 + - owner: prometheus + name: common + languages: + - go + max_files: 800 + - owner: prometheus + name: procfs + languages: + - go + max_files: 1000 + - owner: prometheus + name: docs + languages: + - markdown + max_files: 2000 + - owner: grafana + name: grafana + languages: + - typescript + - go + max_files: 15000 + - owner: grafana + name: loki + languages: + - go + max_files: 8000 + - owner: grafana + name: tempo + languages: + - go + max_files: 3000 + - owner: grafana + name: mimir + languages: + - go + max_files: 4000 + - owner: grafana + name: cortex + languages: + - go + max_files: 3000 + - owner: grafana + name: agent + languages: + - go + max_files: 2500 + - owner: grafana + name: oncall + languages: + - python + max_files: 2500 + - owner: grafana + name: grafana-plugin-sdk-go + languages: + - go + max_files: 1000 + - owner: grafana + name: grafana-plugin-sdk-python + languages: + - python + max_files: 500 + - owner: grafana + name: terraform-provider-grafana + languages: + - go + max_files: 1500 + - owner: grafana + name: helm-charts + languages: + - yaml + max_files: 1500 + - owner: grafana + name: k6 + languages: + - go + - javascript + max_files: 4000 + - owner: grafana + name: xk6 + languages: + - go + max_files: 300 + - owner: grafana + name: xk6-browser + languages: + - go + max_files: 800 + - owner: ansible + name: ansible + languages: + - python + max_files: 15000 + - owner: ansible + name: awx + languages: + - python + - typescript + max_files: 5000 + - owner: ansible + name: ansible-builder + languages: + - python + max_files: 500 + - owner: ansible + name: ansible-compat + languages: + - python + max_files: 300 + - owner: ansible + name: ansible-navigator + languages: + - python + max_files: 1500 + - owner: ansible + name: ansible-runner + languages: + - python + max_files: 1500 + - owner: ansible-collections + name: community.general + languages: + - python + max_files: 1500 + - owner: ansible-collections + name: community.kubernetes + languages: + - python + max_files: 500 + - owner: ansible-collections + name: community.docker + languages: + - python + max_files: 500 + - owner: ansible-collections + name: community.aws + languages: + - python + max_files: 800 + - owner: ansible-collections + name: community.azure + languages: + - python + max_files: 500 + - owner: ansible-collections + name: community.crypto + languages: + - python + max_files: 500 + - owner: ansible-collections + name: community.mysql + languages: + - python + max_files: 400 + - owner: ansible-collections + name: community.postgresql + languages: + - python + max_files: 400 + - owner: ansible-collections + name: community.rabbitmq + languages: + - python + max_files: 300 + - owner: ansible-collections + name: community.redis + languages: + - python + max_files: 400 + - owner: ansible-collections + name: community.mongodb + languages: + - python + max_files: 400 + - owner: ansible-collections + name: community.grafana + languages: + - python + max_files: 500 + - owner: ansible-collections + name: community.hashi_vault + languages: + - python + max_files: 400 + - owner: ansible-collections + name: community.windows + languages: + - powershell + max_files: 500 + - owner: ansible-collections + name: community.network + languages: + - python + max_files: 800 + - owner: ansible-collections + name: community.vmware + languages: + - python + max_files: 800 + - owner: ansible-collections + name: amazon.aws + languages: + - python + max_files: 1500 + - owner: ansible-collections + name: ansible.netcommon + languages: + - python + max_files: 800 + - owner: ansible-collections + name: ansible.posix + languages: + - python + max_files: 300 + - owner: ansible-collections + name: ansible.utils + languages: + - python + max_files: 500 + - owner: ansible-collections + name: ansible.windows + languages: + - powershell + max_files: 500 + - owner: hashicorp + name: terraform-provider-aws + languages: + - go + max_files: 5000 + - owner: hashicorp + name: terraform-provider-google + languages: + - go + max_files: 3000 + - owner: hashicorp + name: terraform-provider-azurerm + languages: + - go + max_files: 4000 + - owner: hashicorp + name: terraform-provider-kubernetes + languages: + - go + max_files: 2000 + - owner: hashicorp + name: terraform-provider-helm + languages: + - go + max_files: 1000 + - owner: hashicorp + name: terraform-provider-vault + languages: + - go + max_files: 2000 + - owner: hashicorp + name: terraform-provider-consul + languages: + - go + max_files: 1000 + - owner: hashicorp + name: terraform-provider-nomad + languages: + - go + max_files: 800 + - owner: hashicorp + name: terraform-provider-null + languages: + - go + max_files: 200 + - owner: hashicorp + name: terraform-provider-random + languages: + - go + max_files: 200 + - owner: hashicorp + name: terraform-provider-local + languages: + - go + max_files: 200 + - owner: hashicorp + name: terraform-provider-tls + languages: + - go + max_files: 300 + - owner: hashicorp + name: terraform-provider-external + languages: + - go + max_files: 200 + - owner: hashicorp + name: terraform-provider-archive + languages: + - go + max_files: 300 + - owner: hashicorp + name: terraform-provider-dns + languages: + - go + max_files: 500 + - owner: hashicorp + name: terraform-provider-http + languages: + - go + max_files: 200 + - owner: hashicorp + name: terraform-provider-gitlab + languages: + - go + max_files: 1000 + - owner: hashicorp + name: terraform-provider-github + languages: + - go + max_files: 1500 + - owner: hashicorp + name: terraform-provider-datadog + languages: + - go + max_files: 1500 + - owner: hashicorp + name: terraform-provider-newrelic + languages: + - go + max_files: 1500 + - owner: hashicorp + name: terraform-provider-snowflake + languages: + - go + max_files: 1500 + - owner: hashicorp + name: terraform-provider-mongodbatlas + languages: + - go + max_files: 1000 + - owner: hashicorp + name: terraform-provider-digitalocean + languages: + - go + max_files: 1500 + - owner: hashicorp + name: terraform-provider-linode + languages: + - go + max_files: 800 + - owner: hashicorp + name: terraform-provider-vultr + languages: + - go + max_files: 500 + - owner: hashicorp + name: terraform-provider-scaleway + languages: + - go + max_files: 800 + - owner: hashicorp + name: terraform-provider-heroku + languages: + - go + max_files: 800 + - owner: hashicorp + name: terraform-provider-pagerduty + languages: + - go + max_files: 800 + - owner: hashicorp + name: terraform-provider-cloudflare + languages: + - go + max_files: 1500 + - owner: hashicorp + name: terraform-provider-fastly + languages: + - go + max_files: 800 + - owner: hashicorp + name: terraform-provider-akamai + languages: + - go + max_files: 800 + - owner: hashicorp + name: terraform-provider-auth0 + languages: + - go + max_files: 800 + - owner: hashicorp + name: terraform-provider-okta + languages: + - go + max_files: 1500 + - owner: hashicorp + name: terraform-provider-azuread + languages: + - go + max_files: 1500 + - owner: hashicorp + name: terraform-provider-azurestack + languages: + - go + max_files: 800 + - owner: hashicorp + name: terraform-provider-azapi + languages: + - go + max_files: 800 + - owner: hashicorp + name: terraform-provider-awscc + languages: + - go + max_files: 1500 + - owner: hashicorp + name: terraform-provider-boundary + languages: + - go + max_files: 800 + - owner: hashicorp + name: terraform-provider-rancher2 + languages: + - go + max_files: 1500 + security: + description: Security (yara, snort, metasploit, nmap) + repos: + - owner: Yara-Rules + name: yara + languages: + - c + max_files: 2500 + - owner: VirusTotal + name: yara + languages: + - c + max_files: 2500 + - owner: snort + name: snort3 + languages: + - c++ + max_files: 3000 + - owner: snort + name: snort + languages: + - c + max_files: 5000 + - owner: OISF + name: suricata + languages: + - c + max_files: 5000 + - owner: OISF + name: libhtp + languages: + - c + max_files: 800 + - owner: rapid7 + name: metasploit-framework + languages: + - ruby + max_files: 8000 + - owner: rapid7 + name: metasploit-omnibus + languages: + - ruby + max_files: 500 + - owner: nmap + name: nmap + languages: + - c + - c++ + - lua + - python + max_files: 5000 + - owner: nmap + name: ncat + languages: + - c + max_files: 800 + - owner: nmap + name: nping + languages: + - c++ + max_files: 800 + - owner: nmap + name: nsock + languages: + - c + max_files: 500 + - owner: nmap + name: libdnet-stripped + languages: + - c + max_files: 300 + - owner: nmap + name: libpcap + languages: + - c + max_files: 1500 + - owner: nmap + name: libpcre + languages: + - c + max_files: 500 + - owner: nmap + name: libssh2 + languages: + - c + max_files: 500 + - owner: nmap + name: libz + languages: + - c + max_files: 500 + - owner: nmap + name: liblua + languages: + - c + max_files: 500 + - owner: nmap + name: nmap-nse-scripts + languages: + - lua + max_files: 1500 + - owner: OpenSCAP + name: openscap + languages: + - c + max_files: 3000 + - owner: OpenSCAP + name: scap-security-guide + languages: + - yaml + max_files: 3000 + - owner: OpenSCAP + name: openscap-daemon + languages: + - python + max_files: 300 + - owner: OpenSCAP + name: scap-workbench + languages: + - c++ + max_files: 800 + - owner: cisagov + name: Malcolm + languages: + - python + max_files: 1500 + - owner: cisagov + name: Cyber-Security + languages: + - python + max_files: 500 + - owner: cisagov + name: ncats-data + languages: + - python + max_files: 300 + - owner: cisagov + name: pshtt + languages: + - python + max_files: 400 + - owner: cisagov + name: trustymail + languages: + - python + max_files: 300 + - owner: CISOfy + name: lynis + languages: + - shell + max_files: 1500 + - owner: aquasecurity + name: kube-bench + languages: + - go + max_files: 1500 + - owner: aquasecurity + name: kube-hunter + languages: + - python + max_files: 800 + - owner: aquasecurity + name: trivy + languages: + - go + max_files: 3000 + - owner: aquasecurity + name: tfsec + languages: + - go + max_files: 2500 + - owner: aquasecurity + name: starboard + languages: + - go + max_files: 1500 + - owner: aquasecurity + name: kubeclarity + languages: + - go + max_files: 1500 + - owner: aquasecurity + name: tracee + languages: + - go + max_files: 2000 + - owner: aquasecurity + name: vulnerability-db + languages: + - go + max_files: 500 + - owner: aquasecurity + name: kubeaudit + languages: + - go + max_files: 800 + - owner: aquasecurity + name: prowler + languages: + - python + max_files: 2000 + ai_tools: + description: AI tooling (LangChain, LlamaIndex, ...) + repos: + - owner: langchain-ai + name: langchain + languages: + - python + - typescript + max_files: 12000 + - owner: langchain-ai + name: langgraph + languages: + - python + max_files: 2500 + - owner: langchain-ai + name: langsmith-sdk + languages: + - python + - typescript + max_files: 1500 + - owner: langchain-ai + name: langchainjs + languages: + - typescript + max_files: 5000 + - owner: langchain-ai + name: langserve + languages: + - python + max_files: 800 + - owner: langchain-ai + name: langchain-experimental + languages: + - python + max_files: 1000 + - owner: langchain-ai + name: langchain-core + languages: + - python + max_files: 2000 + - owner: langchain-ai + name: langchain-community + languages: + - python + max_files: 4000 + - owner: langchain-ai + name: langchain-openai + languages: + - python + max_files: 500 + - owner: langchain-ai + name: langchain-anthropic + languages: + - python + max_files: 400 + - owner: langchain-ai + name: langchain-google-genai + languages: + - python + max_files: 500 + - owner: langchain-ai + name: langchain-google-vertexai + languages: + - python + max_files: 800 + - owner: langchain-ai + name: langchain-aws + languages: + - python + max_files: 800 + - owner: langchain-ai + name: langchain-mistralai + languages: + - python + max_files: 500 + - owner: langchain-ai + name: langchain-cohere + languages: + - python + max_files: 500 + - owner: langchain-ai + name: langchain-huggingface + languages: + - python + max_files: 500 + - owner: langchain-ai + name: langchain-together + languages: + - python + max_files: 300 + - owner: langchain-ai + name: langchain-fireworks + languages: + - python + max_files: 300 + - owner: langchain-ai + name: langchain-groq + languages: + - python + max_files: 300 + - owner: langchain-ai + name: langchain-ollama + languages: + - python + max_files: 400 + - owner: langchain-ai + name: langchain-text-splitters + languages: + - python + max_files: 400 + - owner: run-llama + name: llama_index + languages: + - python + max_files: 8000 + - owner: run-llama + name: llama_index.ts + languages: + - typescript + max_files: 3000 + - owner: run-llama + name: llama_datasets + languages: + - python + max_files: 500 + - owner: explodinggradients + name: ragas + languages: + - python + max_files: 1500 + - owner: microsoft + name: autogen + languages: + - python + max_files: 5000 + - owner: microsoft + name: autogen-typescript + languages: + - typescript + max_files: 1000 + - owner: microsoft + name: FLAML + languages: + - python + max_files: 2000 + - owner: microsoft + name: taskweaver + languages: + - python + max_files: 1500 + - owner: microsoft + name: promptflow + languages: + - python + max_files: 3000 + - owner: crewAIInc + name: crewAI + languages: + - python + max_files: 2500 + - owner: crewAIInc + name: crewAI-tools + languages: + - python + max_files: 500 + - owner: crewAIInc + name: crewAI-examples + languages: + - python + max_files: 800 + - owner: haystack + name: haystack + languages: + - python + max_files: 3000 + - owner: deepset-ai + name: haystack + languages: + - python + max_files: 4000 + - owner: deepset-ai + name: FARM + languages: + - python + max_files: 2000 + - owner: UKPLab + name: sentence-transformers + languages: + - python + max_files: 3000 + - owner: bigscience-workshop + name: promptsource + languages: + - python + max_files: 2000 + - owner: EleutherAI + name: lm-evaluation-harness + languages: + - python + max_files: 2000 + - owner: lmsys + name: FastChat + languages: + - python + max_files: 3000 + - owner: openai + name: openai-cookbook + languages: + - python + - typescript + max_files: 3000 + - owner: openai + name: tiktoken + languages: + - python + - rust + max_files: 800 + - owner: openai + name: whisper + languages: + - python + max_files: 1500 + - owner: openai + name: whisper.cpp + languages: + - c++ + max_files: 1500 + - owner: openai + name: CLIP + languages: + - python + max_files: 1000 + scientific: + description: Scientific (scipy, sympy, astropy, ...) + repos: + - owner: scipy + name: scipy + languages: + - python + - c + - fortran + max_files: 5000 + - owner: numpy + name: numpy + languages: + - python + - c + max_files: 5000 + - owner: matplotlib + name: matplotlib + languages: + - python + max_files: 8000 + - owner: sympy + name: sympy + languages: + - python + max_files: 8000 + - owner: astropy + name: astropy + languages: + - python + - c + max_files: 8000 + - owner: sunpy + name: sunpy + languages: + - python + max_files: 3000 + - owner: biopython + name: biopython + languages: + - python + max_files: 5000 + - owner: openmm + name: openmm + languages: + - c++ + - python + max_files: 3000 + - owner: mdtraj + name: mdtraj + languages: + - python + - c++ + max_files: 2000 + - owner: mdanalysis + name: MDAnalysis + languages: + - python + - c + max_files: 3000 + - owner: openbabel + name: openbabel + languages: + - c++ + max_files: 4000 + - owner: rdkit + name: rdkit + languages: + - c++ + - python + max_files: 5000 + - owner: deepchem + name: deepchem + languages: + - python + max_files: 2500 + - owner: choderalab + name: openmmtools + languages: + - python + max_files: 1500 + - owner: choderalab + name: yank + languages: + - python + max_files: 1000 + - owner: choderalab + name: perses + languages: + - python + max_files: 800 + - owner: openforcefield + name: openff-toolkit + languages: + - python + max_files: 1500 + - owner: openforcefield + name: openff-interchange + languages: + - python + max_files: 800 + - owner: openforcefield + name: openff-evaluator + languages: + - python + max_files: 500 + - owner: MolSSI + name: QCElemental + languages: + - python + max_files: 800 + - owner: MolSSI + name: QCPortal + languages: + - python + max_files: 800 + - owner: MolSSI + name: QCFractal + languages: + - python + max_files: 1500 + - owner: MolSSI + name: QCEngine + languages: + - python + max_files: 1000 + - owner: psi4 + name: psi4 + languages: + - c++ + - python + max_files: 4000 + - owner: psi4 + name: psi4numpy + languages: + - python + max_files: 500 + - owner: geometric + name: geomeTRIC + languages: + - python + max_files: 1000 + - owner: materialsproject + name: pymatgen + languages: + - python + max_files: 5000 + - owner: materialsproject + name: atomate + languages: + - python + max_files: 1500 + - owner: materialsproject + name: atomate2 + languages: + - python + max_files: 1500 + - owner: materialsproject + name: fireworks + languages: + - python + max_files: 1500 + - owner: materialsproject + name: monty + languages: + - python + max_files: 400 + - owner: materialsproject + name: custodian + languages: + - python + max_files: 500 + - owner: materialsproject + name: mp-api + languages: + - python + max_files: 800 + - owner: materialsproject + name: emmet + languages: + - python + max_files: 1000 + - owner: materialsproject + name: jobflow + languages: + - python + max_files: 800 + - owner: materialsproject + name: maggma + languages: + - python + max_files: 1000 + - owner: hackingmaterials + name: matminer + languages: + - python + max_files: 1500 + - owner: hackingmaterials + name: automatminer + languages: + - python + max_files: 800 + - owner: hackingmaterials + name: matbench + languages: + - python + max_files: 500 + systems: + description: Systems (Linux, GNOME, KDE, .NET) + repos: + - owner: torvalds + name: linux + languages: + - c + max_files: 30000 + - owner: golang + name: go + languages: + - go + max_files: 15000 + - owner: rust-lang + name: rust + languages: + - rust + max_files: 20000 + - owner: python + name: cpython + languages: + - python + - c + max_files: 8000 + - owner: python + name: pypy + languages: + - python + - c + max_files: 5000 + - owner: jruby + name: jruby + languages: + - java + - ruby + max_files: 5000 + - owner: ruby + name: ruby + languages: + - c + max_files: 5000 + - owner: php + name: php-src + languages: + - c + max_files: 8000 + - owner: perl + name: perl5 + languages: + - c + max_files: 4000 + - owner: JuliaLang + name: julia + languages: + - julia + - c + max_files: 8000 + - owner: JuliaLang + name: Pkg.jl + languages: + - julia + max_files: 1000 + - owner: JuliaLang + name: Downloads.jl + languages: + - julia + max_files: 300 + - owner: JuliaLang + name: juliaup + languages: + - rust + max_files: 500 + - owner: JuliaLang + name: julia-vscode + languages: + - typescript + max_files: 1000 + - owner: JuliaLang + name: Compat.jl + languages: + - julia + max_files: 200 + - owner: openjdk + name: jdk + languages: + - java + - c + max_files: 15000 + - owner: dotnet + name: runtime + languages: + - c# + - c++ + max_files: 8000 + - owner: dotnet + name: aspnetcore + languages: + - c# + max_files: 10000 + - owner: dotnet + name: efcore + languages: + - c# + max_files: 5000 + - owner: dotnet + name: roslyn + languages: + - c# + max_files: 8000 + - owner: dotnet + name: sdk + languages: + - c# + max_files: 4000 + - owner: dotnet + name: maui + languages: + - c# + max_files: 5000 + - owner: dotnet + name: fsharp + languages: + - f# + max_files: 3000 + - owner: dotnet + name: ilspy + languages: + - c# + max_files: 2000 + - owner: mono + name: mono + languages: + - c + max_files: 5000 + - owner: swig + name: swig + languages: + - c + - python + max_files: 1500 + - owner: gnome + name: glib + languages: + - c + max_files: 5000 + - owner: gnome + name: gtk + languages: + - c + max_files: 8000 + - owner: gnome + name: vala + languages: + - vala + - c + max_files: 3000 + - owner: gnome + name: librsvg + languages: + - rust + max_files: 1500 + - owner: gnome + name: gnome-shell + languages: + - javascript + - c + max_files: 5000 + - owner: gnome + name: mutter + languages: + - c + max_files: 4000 + - owner: gnome + name: gnome-settings-daemon + languages: + - c + max_files: 2000 + - owner: gnome + name: gnome-control-center + languages: + - c + max_files: 3000 + - owner: gnome + name: gnome-software + languages: + - c + max_files: 2500 + - owner: gnome + name: nautilus + languages: + - c + max_files: 3000 + - owner: gnome + name: gnome-boxes + languages: + - vala + max_files: 1500 + - owner: gnome + name: gnome-terminal + languages: + - c + max_files: 1500 + - owner: gnome + name: libsoup + languages: + - c + max_files: 1500 + - owner: gnome + name: gvfs + languages: + - c + max_files: 2000 + - owner: gnome + name: gnome-keyring + languages: + - c + max_files: 2000 + - owner: kde + name: plasma-desktop + languages: + - cpp + max_files: 5000 + - owner: kde + name: plasma-framework + languages: + - cpp + max_files: 2000 + - owner: kde + name: kio + languages: + - cpp + max_files: 2500 + - owner: kde + name: kwin + languages: + - cpp + max_files: 5000 + - owner: kde + name: kirigami + languages: + - cpp + max_files: 1500 + - owner: kde + name: baloo + languages: + - cpp + max_files: 1500 + - owner: kde + name: kdelibs + languages: + - cpp + max_files: 8000 + - owner: kde + name: kdevelop + languages: + - cpp + max_files: 5000 + - owner: kde + name: konsole + languages: + - cpp + max_files: 2000 + - owner: kde + name: okular + languages: + - cpp + max_files: 3000 + - owner: kde + name: dolphin + languages: + - cpp + max_files: 2000 + - owner: kde + name: kate + languages: + - cpp + max_files: 3000 diff --git a/configs/nexus_coder_10b.yaml b/configs/nexus_coder_10b.yaml new file mode 100644 index 0000000000000000000000000000000000000000..399bd48d26f02ccbac84a2a8d649b0f5fc947bd3 --- /dev/null +++ b/configs/nexus_coder_10b.yaml @@ -0,0 +1,120 @@ +# Nexus Coder Configuration - Large (10B/1.5B) v0.3 - DEFAULT +# Author: Hieu Louis (2026) +# Default model. 32+ GPU recommended for full pretrain. + +model: + name: "Nexus Coder" + agent_name: "Nexus" + author: "Hieu Louis" + version: "0.3.0" + github: "mhieuhonda" + year: "2026" + +architecture: + vocab_size: 32000 + hidden_size: 2048 + num_hidden_layers: 12 + num_attention_heads: 16 + num_kv_heads: 4 # Grouped Query Attention + head_dim: 128 + intermediate_size: 5632 # per-expert + hidden_act: "silu" # SwiGLU + norm_type: "rmsnorm" + +moe: + num_experts: 24 # Tổng số chuyên gia + num_active_experts: 3 # Chuyên gia kích hoạt mỗi token + router_aux_loss_coef: 0.001 + router_jitter_noise: 0.0 + +context: + max_position_embeddings: 50000 # 50k tokens + rotary_emb_base: 10000.0 + rope_scaling_type: null + rope_scaling_factor: 1.0 + +# v0.3 NEW attention features +attention: + use_flash_attention: true # PyTorch SDPA + use_flash_attention_2: false # FlashAttention-2 (optional, install flash-attn) + use_alibi: false # ALiBi alternative to RoPE + alibi_max_slope: 8.0 + use_sliding_window: true # alternating SWA / global layers + sliding_window_size: 4096 + sliding_window_layers: null # null = alternate even/odd layers + use_qk_norm: true # RMSNorm on Q and K (Llama-3 style) + qk_norm_eps: 1.0e-6 + mlp_parallel: true # fused gate+up projection + +compute: + use_kv_cache: true + kv_cache_quantization: null # null | "int8" | "fp8" + gradient_checkpointing: false + tensor_parallel_size: 1 + pipeline_parallel_size: 1 + expert_parallel_size: 1 + sequence_parallel: false + +params: + total: "~10.22B" + active: "~1.50B" + expert_utilization: "12.5%" + estimated_disk_mb_fp16: 19500 + estimated_disk_mb_int8: 9750 + estimated_disk_mb_int4: 4875 + +training: + learning_rate: 5.0e-4 + weight_decay: 0.01 + warmup_steps: 100 + max_steps: 5000 + per_device_batch_size: 4 + gradient_accumulation_steps: 4 + logging_steps: 10 + save_steps: 500 + max_grad_norm: 1.0 + seed: 42 + use_amp: true + +inference: + max_new_tokens: 200 + temperature: 0.8 + top_k: 50 + top_p: 0.9 + do_sample: true + +personality: + type: "humorous" + language: "bilingual" + specialties: + - programming + - conversation + - devops + - ml + - security + +# v0.3 NEW capabilities +capabilities: + skills_count: 60 + tools_count: 80 + data_sources: + - github + - huggingface + - arxiv + - wikipedia + - stackoverflow + - the_stack + - starcoder2_data + - python_alpaca + training_frameworks_referenced: + - litgpt + - llamafactory + - axolotl + - openhands + - omp_gym + +environment: + python_version: "3.12.13" + pytorch_version: ">=2.0" + cuda_required: false # có thể chạy trên CPU (chậm) + recommended_gpus: "32+ H100 80GB for full pretrain" diff --git a/configs/nexus_coder_30b.yaml b/configs/nexus_coder_30b.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e2bab8035a9267e354c7a0d18b30aab93a71025e --- /dev/null +++ b/configs/nexus_coder_30b.yaml @@ -0,0 +1,103 @@ +# Nexus Coder Configuration - 30B/3B v0.3 NEW +# Pretrain on 64-128 H100 80GB GPUs. +# Recommended for serious pretraining at frontier scale. +# Author: Hieu Louis (2026) + +model: + name: "Nexus Coder 30B" + agent_name: "Nexus" + author: "Hieu Louis" + version: "0.3.0-30b" + github: "mhieuhonda" + year: "2026" + +architecture: + vocab_size: 64000 + hidden_size: 4096 + num_hidden_layers: 24 + num_attention_heads: 32 + num_kv_heads: 8 + head_dim: 128 + intermediate_size: 11264 + hidden_act: "silu" + norm_type: "rmsnorm" + +moe: + num_experts: 48 + num_active_experts: 4 + router_aux_loss_coef: 0.001 + router_jitter_noise: 0.0 + +context: + max_position_embeddings: 65536 + rotary_emb_base: 10000.0 + rope_scaling_type: "dynamic" # NTK-aware scaling for 2× context + rope_scaling_factor: 2.0 + +attention: + use_flash_attention: true + use_flash_attention_2: true # mandatory at this scale + use_alibi: false + use_sliding_window: true + sliding_window_size: 8192 + use_qk_norm: true + qk_norm_eps: 1.0e-6 + mlp_parallel: true + +compute: + use_kv_cache: true + kv_cache_quantization: "int8" + gradient_checkpointing: true + tensor_parallel_size: 4 + pipeline_parallel_size: 1 + expert_parallel_size: 4 + sequence_parallel: false + +params: + total: "~30B" + active: "~3B" + expert_utilization: "8.3%" + estimated_disk_mb_fp16: 60000 + estimated_disk_mb_int8: 30000 + estimated_disk_mb_int4: 15000 + kv_cache_mb_per_token_fp16: 0.019 + kv_cache_mb_per_token_int8: 0.0095 + +training: + learning_rate: 2.0e-4 + weight_decay: 0.01 + warmup_steps: 500 + max_steps: 10000 + per_device_batch_size: 1 + gradient_accumulation_steps: 32 + logging_steps: 10 + save_steps: 1000 + max_grad_norm: 1.0 + seed: 42 + use_amp: true + use_deepspeed: true + deepspeed_config: "configs/ds_config_zero3.json" + total_tokens_target: 500_000_000_000 # 500B tokens + +inference: + max_new_tokens: 1000 + temperature: 0.7 + top_k: 50 + top_p: 0.9 + do_sample: true + +personality: + type: "humorous" + language: "bilingual" + +capabilities: + skills_count: 60 + tools_count: 80 + +environment: + python_version: "3.12.13" + pytorch_version: ">=2.0" + cuda_required: true + min_gpu_memory_gb: 80 + recommended_gpus: "64-128 H100 80GB" + estimated_training_time: "~30 days on 64 H100s" diff --git a/configs/nexus_coder_423b.yaml b/configs/nexus_coder_423b.yaml new file mode 100644 index 0000000000000000000000000000000000000000..71c64673c74a5e90e18d01a7f6390536fcfe938f --- /dev/null +++ b/configs/nexus_coder_423b.yaml @@ -0,0 +1,89 @@ +# ============================================================================ +# Nexus Coder v0.4 — CyberForge Config (423B / 39B / 3M context) +# ============================================================================ +# Supreme variant — CyberGym training hooks enabled by default. +# Math (verified): +# embed (200k × 7168) = 1.43B +# per_layer_total (48 exp) = 17.03B +# per_layer_active (4 exp) = 1.53B +# 24 layers = 408B total / 36.6B active +# + LM head + norms + routers = ~412-423B total / ~39.5B active +# ============================================================================ +# Recommended hardware: +# - 8× H100 80GB (TP=8) or 16× A100 80GB (TP=8, EP=2) +# - ~600 GB RAM for data loading +# - 3M context requires gradient checkpointing + KV int8 cache +# ============================================================================ + +name: "Nexus Coder 423B" +version: "0.4.0" +author: "Hieu Louis" + +# === Architecture === +vocab_size: 200000 +hidden_size: 7168 +num_hidden_layers: 24 +num_attention_heads: 56 +num_kv_heads: 8 +head_dim: 128 +intermediate_size: 16384 +hidden_act: "silu" +num_experts: 48 +num_active_experts: 4 +router_aux_loss_coef: 0.001 + +# === Context window (3M tokens via YaRN ×60) === +max_position_embeddings: 3000000 +rotary_emb_base: 1000000.0 # larger base for long context +rope_scaling_type: "yarn" +rope_scaling_factor: 60.0 +yarn_beta_fast: 32.0 +yarn_beta_slow: 1.0 + +# === Attention features === +use_flash_attention: true +use_flash_attention_2: true +use_qk_norm: true +qk_norm_eps: 1.0e-6 +mlp_parallel: true +use_sliding_window: true +sliding_window_size: 32768 +use_alibi: false + +# === Memory optimizations === +gradient_checkpointing: true +kv_cache_quantization: "int8" +kv_cache_bits: 8 + +# === v0.4 CyberGym === +cybergym_enabled: true +cybergym_mutation_rate: 0.01 +cybergym_mutation_sigma: 1.0e-4 +cybergym_mutation_period: 500 +cybergym_keep_ratio: 0.7 +cybergym_adaptive_routing: true +cybergym_min_active_experts: 2 +cybergym_max_active_experts: 8 +cybergym_genome_init: true +cybergym_cep_stages: [32768, 131072, 524288, 1048576, 2097152, 3000000] +cybergym_cep_epoch_per_stage: 1 + +# === Distributed === +tensor_parallel_size: 8 +pipeline_parallel_size: 1 +expert_parallel_size: 8 +sequence_parallel: false + +# === Training defaults === +pad_token_id: 0 +bos_token_id: 1 +eos_token_id: 2 +unk_token_id: 3 + +# === Safety === +enable_safety_filter: true +max_output_tokens: 8192 + +# === Personality === +personality: "humorous" +language: "bilingual" diff --git a/configs/nexus_coder_70b.yaml b/configs/nexus_coder_70b.yaml new file mode 100644 index 0000000000000000000000000000000000000000..11d060abc1135ebd13fcdf23382fd37a5e4616eb --- /dev/null +++ b/configs/nexus_coder_70b.yaml @@ -0,0 +1,106 @@ +# Nexus Coder Configuration - 70B/5B v0.3 NEW (frontier research) +# Author: Hieu Louis (2026) +# RESEARCH ONLY. Requires 256+ H100/H200 GPUs or equivalent. +# Uses YaRN RoPE scaling for 4× context extension → 128k tokens. + +model: + name: "Nexus Coder 70B" + agent_name: "Nexus" + author: "Hieu Louis" + version: "0.3.0-70b" + github: "mhieuhonda" + year: "2026" + +architecture: + vocab_size: 128000 # tiktoken-style tokenizer + hidden_size: 6144 + num_hidden_layers: 32 + num_attention_heads: 48 + num_kv_heads: 8 # heavy GQA (6:1 ratio) + head_dim: 128 + intermediate_size: 16384 + hidden_act: "silu" + norm_type: "rmsnorm" + +moe: + num_experts: 64 # Frontier-scale MoE + num_active_experts: 4 + router_aux_loss_coef: 0.001 + router_jitter_noise: 0.0 + +context: + max_position_embeddings: 131072 # 128k tokens + rotary_emb_base: 500000.0 # larger base for long context + rope_scaling_type: "yarn" # YaRN — SOTA for 4×+ extension + rope_scaling_factor: 4.0 + yarn_beta_fast: 32.0 + yarn_beta_slow: 1.0 + +attention: + use_flash_attention: true + use_flash_attention_2: true + use_alibi: false # YaRN handles long context + use_sliding_window: true + sliding_window_size: 16384 + use_qk_norm: true + qk_norm_eps: 1.0e-6 + mlp_parallel: true + +compute: + use_kv_cache: true + kv_cache_quantization: "fp8" # FP8 KV cache for memory efficiency + gradient_checkpointing: true + tensor_parallel_size: 8 + pipeline_parallel_size: 2 + expert_parallel_size: 8 + sequence_parallel: true # enable sequence parallel for long context + +params: + total: "~70B" + active: "~5B" + expert_utilization: "6.25%" + estimated_disk_mb_fp16: 140000 + estimated_disk_mb_int8: 70000 + estimated_disk_mb_int4: 35000 + kv_cache_mb_per_token_fp16: 0.050 + kv_cache_mb_per_token_fp8: 0.025 + +training: + learning_rate: 1.5e-4 + weight_decay: 0.01 + warmup_steps: 2000 + max_steps: 50000 + per_device_batch_size: 1 + gradient_accumulation_steps: 128 + logging_steps: 10 + save_steps: 2000 + max_grad_norm: 1.0 + seed: 42 + use_amp: true + use_deepspeed: true + deepspeed_config: "configs/ds_config_zero3_offload.json" + total_tokens_target: 1_500_000_000_000 # 1.5T tokens + +inference: + max_new_tokens: 2000 + temperature: 0.7 + top_k: 50 + top_p: 0.9 + do_sample: true + +personality: + type: "humorous" + language: "bilingual" + +capabilities: + skills_count: 60 + tools_count: 80 + +environment: + python_version: "3.12.13" + pytorch_version: ">=2.3" + cuda_required: true + min_gpu_memory_gb: 80 + recommended_gpus: "256+ H100/H200 80GB" + estimated_training_time: "~90 days on 256 H100s" + notes: "This config is research-only. Use 30B or 10B for production." diff --git a/configs/nexus_coder_medium.yaml b/configs/nexus_coder_medium.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5ac82a62d18d4918b1b99549c6f06fcdf6e1001a --- /dev/null +++ b/configs/nexus_coder_medium.yaml @@ -0,0 +1,80 @@ +# Nexus Coder Configuration - Medium version v0.3 +# ~1B params, pretrain on 4-8 GPU +# Author: Hieu Louis (2026) + +model: + name: "Nexus Coder Medium" + agent_name: "Nexus" + author: "Hieu Louis" + version: "0.3.0-medium" + github: "mhieuhonda" + year: "2026" + +architecture: + vocab_size: 32000 + hidden_size: 1536 + num_hidden_layers: 24 + num_attention_heads: 16 + num_kv_heads: 4 + head_dim: 96 + intermediate_size: 4096 + hidden_act: "silu" + norm_type: "rmsnorm" + +moe: + num_experts: 16 + num_active_experts: 2 + router_aux_loss_coef: 0.001 + +context: + max_position_embeddings: 16384 + rotary_emb_base: 10000.0 + +attention: + use_flash_attention: true + use_flash_attention_2: false + use_alibi: false + use_sliding_window: true + sliding_window_size: 2048 + use_qk_norm: true + mlp_parallel: true + +compute: + use_kv_cache: true + kv_cache_quantization: null + gradient_checkpointing: false + +params: + total: "~1.1B" + active: "~250M" + expert_utilization: "12.5%" + +training: + learning_rate: 3.0e-4 + weight_decay: 0.01 + warmup_steps: 100 + max_steps: 5000 + per_device_batch_size: 4 + gradient_accumulation_steps: 4 + logging_steps: 10 + save_steps: 500 + max_grad_norm: 1.0 + seed: 42 + use_amp: true + +inference: + max_new_tokens: 200 + temperature: 0.8 + top_k: 50 + top_p: 0.9 + do_sample: true + +personality: + type: "humorous" + language: "bilingual" + +environment: + python_version: "3.12.13" + pytorch_version: ">=2.0" + cuda_required: true + min_gpu_memory_gb: 16 diff --git a/configs/nexus_coder_small.yaml b/configs/nexus_coder_small.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ab484c94bb905b016857050b2c0dd700f2e3e703 --- /dev/null +++ b/configs/nexus_coder_small.yaml @@ -0,0 +1,77 @@ +# Nexus Coder Configuration - Small version v0.3 +# ~125M params, fine-tune on 1 GPU +# Author: Hieu Louis (2026) + +model: + name: "Nexus Coder Small" + agent_name: "Nexus" + author: "Hieu Louis" + version: "0.3.0-small" + github: "mhieuhonda" + year: "2026" + +architecture: + vocab_size: 16000 + hidden_size: 768 + num_hidden_layers: 12 + num_attention_heads: 12 + num_kv_heads: 4 + head_dim: 64 + intermediate_size: 2048 + hidden_act: "silu" + norm_type: "rmsnorm" + +moe: + num_experts: 8 + num_active_experts: 2 + router_aux_loss_coef: 0.001 + +context: + max_position_embeddings: 8192 + rotary_emb_base: 10000.0 + +attention: + use_flash_attention: true + use_flash_attention_2: false + use_alibi: false + use_sliding_window: false + use_qk_norm: true + mlp_parallel: true + +compute: + use_kv_cache: true + kv_cache_quantization: null + gradient_checkpointing: false + +params: + total: "~125M" + active: "~45M" + expert_utilization: "25%" + +training: + learning_rate: 3.0e-4 + weight_decay: 0.01 + warmup_steps: 50 + max_steps: 1000 + per_device_batch_size: 8 + gradient_accumulation_steps: 2 + logging_steps: 10 + save_steps: 200 + max_grad_norm: 1.0 + seed: 42 + +inference: + max_new_tokens: 200 + temperature: 0.8 + top_k: 50 + top_p: 0.9 + do_sample: true + +personality: + type: "humorous" + language: "bilingual" + +environment: + python_version: "3.12.13" + pytorch_version: ">=2.0" + cuda_required: false diff --git a/configs/nexus_coder_tiny.yaml b/configs/nexus_coder_tiny.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6c6eefb10f3aa98b7dbf55a5ab4ff06fe9fe2a6e --- /dev/null +++ b/configs/nexus_coder_tiny.yaml @@ -0,0 +1,80 @@ +# Nexus Coder Configuration - Tiny version v0.3 +# Used for quick verification on CPU (~5M params) +# Author: Hieu Louis (2026) + +model: + name: "Nexus Coder Tiny" + agent_name: "Nexus" + author: "Hieu Louis" + version: "0.3.0-tiny" + github: "mhieuhonda" + year: "2026" + +architecture: + vocab_size: 2000 + hidden_size: 256 + num_hidden_layers: 4 + num_attention_heads: 8 + num_kv_heads: 2 + head_dim: 32 + intermediate_size: 512 + hidden_act: "silu" + norm_type: "rmsnorm" + +moe: + num_experts: 4 + num_active_experts: 2 + router_aux_loss_coef: 0.001 + router_jitter_noise: 0.0 + +context: + max_position_embeddings: 512 + rotary_emb_base: 10000.0 + rope_scaling_type: null + rope_scaling_factor: 1.0 + +# v0.3 NEW architecture features (most OFF for tiny — too small to benefit) +attention: + use_flash_attention: false + use_flash_attention_2: false + use_alibi: false + use_sliding_window: false + sliding_window_size: 256 + use_qk_norm: false + mlp_parallel: true + +compute: + use_kv_cache: true + kv_cache_quantization: null + gradient_checkpointing: false + +params: + total: "~8M (demo only)" + active: "~5M" + note: "For testing only. Use nexus_coder_10b.yaml for the real model." + +training: + learning_rate: 5.0e-4 + weight_decay: 0.01 + warmup_steps: 10 + max_steps: 30 + per_device_batch_size: 2 + gradient_accumulation_steps: 1 + logging_steps: 5 + save_steps: 30 + +inference: + max_new_tokens: 50 + temperature: 0.8 + top_k: 50 + top_p: 0.9 + do_sample: true + +personality: + type: "humorous" + language: "bilingual" + +environment: + python_version: "3.12.13" + pytorch_version: ">=2.0" + cuda_required: false diff --git a/configs/nexus_coder_xlarge.yaml b/configs/nexus_coder_xlarge.yaml new file mode 100644 index 0000000000000000000000000000000000000000..99aaddc93c774dd1b2ed9e42de87f5663e44cef4 --- /dev/null +++ b/configs/nexus_coder_xlarge.yaml @@ -0,0 +1,92 @@ +# Nexus Coder Configuration - XLarge (~30B/3B) v0.3 +# Research-only. Requires 64+ H100 80GB GPUs. +# Author: Hieu Louis (2026) + +model: + name: "Nexus Coder XLarge" + agent_name: "Nexus" + author: "Hieu Louis" + version: "0.3.0-xlarge" + github: "mhieuhonda" + year: "2026" + +architecture: + vocab_size: 64000 + hidden_size: 4096 + num_hidden_layers: 24 + num_attention_heads: 32 + num_kv_heads: 8 + head_dim: 128 + intermediate_size: 11264 + hidden_act: "silu" + norm_type: "rmsnorm" + +moe: + num_experts: 48 + num_active_experts: 4 + router_aux_loss_coef: 0.001 + +context: + max_position_embeddings: 65536 # 64k tokens + rotary_emb_base: 10000.0 + rope_scaling_type: "dynamic" # NTK-aware for 2× context extension + rope_scaling_factor: 2.0 + +attention: + use_flash_attention: true + use_flash_attention_2: true # recommended at this scale + use_alibi: false + use_sliding_window: true + sliding_window_size: 8192 + use_qk_norm: true + mlp_parallel: true + +compute: + use_kv_cache: true + kv_cache_quantization: "int8" # saves KV cache memory at long context + gradient_checkpointing: true # essential at this scale + tensor_parallel_size: 4 + pipeline_parallel_size: 1 + expert_parallel_size: 4 + sequence_parallel: false + +params: + total: "~30B" + active: "~3B" + expert_utilization: "8.3%" + estimated_disk_mb_fp16: 60000 + estimated_disk_mb_int8: 30000 + estimated_disk_mb_int4: 15000 + +training: + learning_rate: 2.0e-4 + weight_decay: 0.01 + warmup_steps: 500 + max_steps: 10000 + per_device_batch_size: 1 + gradient_accumulation_steps: 32 + logging_steps: 10 + save_steps: 1000 + max_grad_norm: 1.0 + seed: 42 + use_amp: true + use_deepspeed: true + deepspeed_config: "configs/ds_config_zero3.json" + +inference: + max_new_tokens: 500 + temperature: 0.7 + top_k: 50 + top_p: 0.9 + do_sample: true + +personality: + type: "humorous" + language: "bilingual" + +environment: + python_version: "3.12.13" + pytorch_version: ">=2.0" + cuda_required: true + min_gpu_memory_gb: 80 + recommended_gpus: "64+ H100 80GB" diff --git a/configs/sources.yaml b/configs/sources.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2f16a96bda234122d6c5d83cda3e55a40c6852e3 --- /dev/null +++ b/configs/sources.yaml @@ -0,0 +1,685 @@ +# Nexus Coder v0.3 - Training Data Sources Configuration +# ======================================================= +# Curated sources for pre-training Nexus Coder v0.3. +# Target: ~500 GitHub repos + ~150 HuggingFace datasets + 5 new sources. +# +# Author: Hieu Louis (2026) +# Total estimated tokens (post-filtering): ~50B-200B +# +# References (the inspiration for many of these sources): +# - litgpt's curated pretraining datasets +# - LlamaFactory's example configs +# - axolotl's dataset registry +# - StarCoder2 paper data card +# - The-Stack v2 dataset card + +# ============================================================================= +# GitHub — curated code repos (~500) +# ============================================================================= +github: + enabled: true + cache_dir: "./data_cache/github" + max_concurrent: 8 + max_files_per_repo: 1000 + max_file_size_kb: 100 + + repos: + # ---------- Python core & stdlib (5) ---------- + - {owner: "python", name: "cpython", languages: ["python"], max_files: 3000} + - {owner: "pallets", name: "flask", languages: ["python"]} + - {owner: "pallets", name: "django", languages: ["python"], max_files: 2000} + - {owner: "psf", name: "requests", languages: ["python"]} + - {owner: "pallets", name: "click", languages: ["python"]} + + # ---------- Python: data science (8) ---------- + - {owner: "numpy", name: "numpy", languages: ["python"], max_files: 2000} + - {owner: "pandas-dev", name: "pandas", languages: ["python"], max_files: 2000} + - {owner: "scipy", name: "scipy", languages: ["python"], max_files: 2000} + - {owner: "matplotlib", name: "matplotlib", languages: ["python"], max_files: 2000} + - {owner: "scikit-learn", name: "scikit-learn", languages: ["python"], max_files: 2000} + - {owner: "plotly", name: "plotly.py", languages: ["python"]} + - {owner: "bokeh", name: "bokeh", languages: ["python"]} + - {owner: "sympy", name: "sympy", languages: ["python"], max_files: 2000} + + # ---------- Python: ML / DL (10) ---------- + - {owner: "pytorch", name: "pytorch", languages: ["python", "cpp"], max_files: 3000} + - {owner: "tensorflow", name: "tensorflow", languages: ["python", "cpp"], max_files: 3000} + - {owner: "huggingface", name: "transformers", languages: ["python"], max_files: 3000} + - {owner: "huggingface", name: "datasets", languages: ["python"]} + - {owner: "huggingface", name: "peft", languages: ["python"]} + - {owner: "huggingface", name: "accelerate", languages: ["python"]} + - {owner: "huggingface", name: "tokenizers", languages: ["python", "rust"]} + - {owner: "langchain-ai", name: "langchain", languages: ["python"], max_files: 2000} + - {owner: "run-llama", name: "llama_index", languages: ["python"]} + - {owner: "explosion", name: "spaCy", languages: ["python"]} + + # ---------- Python: Web frameworks (8) ---------- + - {owner: "tiangolo", name: "fastapi", languages: ["python"], max_files: 2000} + - {owner: "encode", name: "starlette", languages: ["python"]} + - {owner: "encode", name: "uvicorn", languages: ["python"]} + - {owner: "django", name: "djangoproject.com", languages: ["python"]} + - {owner: "falconry", name: "falcon", languages: ["python"]} + - {owner: "sanic-org", name: "sanic", languages: ["python"]} + - {owner: "tornadoweb", name: "tornado", languages: ["python"]} + - {owner: "aio-libs", name: "aiohttp", languages: ["python"]} + + # ---------- Python: tools (8) ---------- + - {owner: "pytest-dev", name: "pytest", languages: ["python"]} + - {owner: "psf", name: "black", languages: ["python"]} + - {owner: "pydantic", name: "pydantic", languages: ["python"]} + - {owner: "pypa", name: "pip", languages: ["python"]} + - {owner: "pypa", name: "setuptools", languages: ["python"]} + - {owner: "pyca", name: "cryptography", languages: ["python", "c"]} + - {owner: "celery", name: "celery", languages: ["python"]} + - {owner: "mwclient", name: "redis-py", languages: ["python"]} + + # ---------- Python: async / networking (5) ---------- + - {owner: "aio-libs", name: "aiomysql", languages: ["python"]} + - {owner: "MagicStack", name: "asyncpg", languages: ["python", "cython"]} + - {owner: "sqlalchemy", name: "sqlalchemy", languages: ["python"], max_files: 2000} + - {owner: "scrapy", name: "scrapy", languages: ["python"]} + - {owner: "httpx", name: "httpx", languages: ["python"]} + + # ---------- Python: DevOps / Infra (5) ---------- + - {owner: "ansible", name: "ansible", languages: ["python"], max_files: 2000} + - {owner: "openstack", name: "openstack", languages: ["python"]} + - {owner: "saltstack", name: "salt", languages: ["python"]} + - {owner: "aws", name: "aws-cli", languages: ["python"]} + - {owner: "boto", name: "boto3", languages: ["python"]} + + # ---------- JavaScript / TypeScript (10) ---------- + - {owner: "facebook", name: "react", languages: ["javascript", "typescript"], max_files: 2000} + - {owner: "vuejs", name: "vue", languages: ["javascript", "typescript"], max_files: 2000} + - {owner: "vercel", name: "next.js", languages: ["javascript", "typescript"], max_files: 2000} + - {owner: "angular", name: "angular", languages: ["typescript"], max_files: 2000} + - {owner: "sveltejs", name: "svelte", languages: ["javascript", "typescript"]} + - {owner: "microsoft", name: "TypeScript", languages: ["typescript"], max_files: 3000} + - {owner: "nodejs", name: "node", languages: ["javascript", "cpp"], max_files: 2000} + - {owner: "denoland", name: "deno", languages: ["typescript", "rust"], max_files: 2000} + - {owner: "expressjs", name: "express", languages: ["javascript"]} + - {owner: "fastify", name: "fastify", languages: ["javascript"]} + + # ---------- JavaScript / TypeScript: tools (5) ---------- + - {owner: "eslint", name: "eslint", languages: ["javascript"]} + - {owner: "prettier", name: "prettier", languages: ["javascript", "typescript"]} + - {owner: "webpack", name: "webpack", languages: ["javascript"], max_files: 2000} + - {owner: "vitejs", name: "vite", languages: ["typescript"]} + - {owner: "rollup", name: "rollup", languages: ["javascript", "typescript"]} + + # ---------- Go (10) ---------- + - {owner: "golang", name: "go", languages: ["go"], max_files: 3000} + - {owner: "gin-gonic", name: "gin", languages: ["go"]} + - {owner: "kubernetes", name: "kubernetes", languages: ["go"], max_files: 3000} + - {owner: "prometheus", name: "prometheus", languages: ["go"], max_files: 2000} + - {owner: "hashicorp", name: "terraform", languages: ["go"], max_files: 2000} + - {owner: "hashicorp", name: "consul", languages: ["go"]} + - {owner: "hashicorp", name: "vault", languages: ["go"]} + - {owner: "etcd-io", name: "etcd", languages: ["go"]} + - {owner: "docker", name: "compose", languages: ["go"]} + - {owner: "gohugoio", name: "hugo", languages: ["go"]} + + # ---------- Go: more tools (5) ---------- + - {owner: "spf13", name: "cobra", languages: ["go"]} + - {owner: "spf13", name: "viper", languages: ["go"]} + - {owner: "golang", name: "mock", languages: ["go"]} + - {owner: "stretchr", name: "testify", languages: ["go"]} + - {owner: "grpc", name: "grpc-go", languages: ["go"]} + + # ---------- Rust (10) ---------- + - {owner: "rust-lang", name: "rust", languages: ["rust"], max_files: 3000} + - {owner: "tokio-rs", name: "tokio", languages: ["rust"], max_files: 2000} + - {owner: "serde-rs", name: "serde", languages: ["rust"]} + - {owner: "BurntSushi", name: "ripgrep", languages: ["rust"]} + - {owner: "sharkdp", name: "bat", languages: ["rust"]} + - {owner: "sharkdp", name: "fd", languages: ["rust"]} + - {owner: "BurntSushi", name: "csv", languages: ["rust"]} + - {owner: "rust-lang", name: "cargo", languages: ["rust"]} + - {owner: "rust-lang", name: "rustfmt", languages: ["rust"]} + - {owner: "delta-io", name: "delta-rs", languages: ["rust"]} + + # ---------- Rust: web / async (5) ---------- + - {owner: "actix", name: "actix-web", languages: ["rust"]} + - {owner: "axo", name: "axum", languages: ["rust"]} + - {owner: "hyperium", name: "hyper", languages: ["rust"]} + - {owner: "hyperium", name: "tonic", languages: ["rust"]} + - {owner: "seanmonstar", name: "reqwest", languages: ["rust"]} + + # ---------- C / C++ (8) ---------- + - {owner: "llvm", name: "llvm-project", languages: ["cpp"], max_files: 3000} + - {owner: "gcc-mirror", name: "gcc", languages: ["cpp", "c"], max_files: 2000} + - {owner: "cmake", name: "cmake", languages: ["cpp"]} + - {owner: "google", name: "googletest", languages: ["cpp"]} + - {owner: "fmtlib", name: "fmt", languages: ["cpp"]} + - {owner: "gabime", name: "spdlog", languages: ["cpp"]} + - {owner: "nlohmann", name: "json", languages: ["cpp"]} + - {owner: "grpc", name: "grpc", languages: ["cpp", "c"], max_files: 2000} + + # ---------- Java (6) ---------- + - {owner: "spring-projects", name: "spring-boot", languages: ["java"], max_files: 2000} + - {owner: "apache", name: "kafka", languages: ["java", "scala"], max_files: 2000} + - {owner: "apache", name: "cassandra", languages: ["java"]} + - {owner: "apache", name: "maven", languages: ["java"]} + - {owner: "apache", name: "tomcat", languages: ["java"]} + - {owner: "OpenLiberty", name: "open-liberty", languages: ["java"]} + + # ---------- Java: tools (4) ---------- + - {owner: "junit-team", name: "junit5", languages: ["java"]} + - {owner: "mockito", name: "mockito", languages: ["java"]} + - {owner: "GoogleJavaFormat", name: "google-java-format", languages: ["java"]} + - {owner: "checkstyle", name: "checkstyle", languages: ["java"]} + + # ---------- C# / .NET (4) ---------- + - {owner: "dotnet", name: "aspnetcore", languages: ["c#"], max_files: 2000} + - {owner: "dotnet", name: "runtime", languages: ["c#"], max_files: 2000} + - {owner: "dotnet", name: "efcore", languages: ["c#"]} + - {owner: "dotnet", name: "roslyn", languages: ["c#"], max_files: 2000} + + # ---------- Ruby (3) ---------- + - {owner: "rails", name: "rails", languages: ["ruby"], max_files: 2000} + - {owner: "ruby", name: "ruby", languages: ["c", "ruby"], max_files: 2000} + - {owner: "sinatra", name: "sinatra", languages: ["ruby"]} + + # ---------- PHP (3) ---------- + - {owner: "laravel", name: "framework", languages: ["php"], max_files: 2000} + - {owner: "symfony", name: "symfony", languages: ["php"], max_files: 2000} + - {owner: "php", name: "php-src", languages: ["c"], max_files: 2000} + + # ---------- Swift (2) ---------- + - {owner: "apple", name: "swift", languages: ["swift"], max_files: 2000} + - {owner: "vapor", name: "vapor", languages: ["swift"]} + + # ---------- Kotlin (3) ---------- + - {owner: "JetBrains", name: "kotlin", languages: ["kotlin"], max_files: 2000} + - {owner: "Kotlin", name: "ktor", languages: ["kotlin"]} + - {owner: "android", name: "architecture-components-samples", languages: ["kotlin"]} + + # ---------- ML / DL / LLM (extra, 8) ---------- + - {owner: "stanfordnlp", name: "stanford-alpaca", languages: ["python"]} + - {owner: "tatsu-lab", name: "stanford_alpaca", languages: ["python"]} + - {owner: "lm-sys", name: "FastChat", languages: ["python"]} + - {owner: "OpenAccess-AI-Collective", name: "axolotl", languages: ["python"]} + - {owner: "Lightning-AI", name: "litgpt", languages: ["python"]} + - {owner: "hiyouga", name: "LLaMA-Factory", languages: ["python"]} + - {owner: "vllm-project", name: "vllm", languages: ["python", "cpp"], max_files: 2000} + - {owner: "sgl-project", name: "sglang", languages: ["python", "cpp"]} + + # ---------- AI agents (5) ---------- + - {owner: "OpenHands", name: "OpenHands", languages: ["python"]} + - {owner: "langchain-ai", name: "langgraph", languages: ["python"]} + - {owner: "crewAIInc", name: "crewAI", languages: ["python"]} + - {owner: "microsoft", name: "autogen", languages: ["python"]} + - {owner: "openai", name: "openai-python", languages: ["python"]} + + # ---------- DevOps / Infrastructure (8) ---------- + - {owner: "docker", name: "docker-ce", languages: ["go"], max_files: 2000} + - {owner: "containerd", name: "containerd", languages: ["go"]} + - {owner: "opencontainers", name: "image-spec", languages: ["go"]} + - {owner: "cncf", name: "landscape", languages: ["yaml"]} + - {owner: "helm", name: "helm", languages: ["go"]} + - {owner: "istio", name: "istio", languages: ["go"], max_files: 2000} + - {owner: "envoyproxy", name: "envoy", languages: ["cpp"], max_files: 2000} + - {owner: "traefik", name: "traefik", languages: ["go"]} + + # ---------- Database / Storage (5) ---------- + - {owner: "postgres", name: "postgres", languages: ["c"], max_files: 2000} + - {owner: "mysql", name: "mysql-server", languages: ["cpp"], max_files: 2000} + - {owner: "sqlite", name: "sqlite", languages: ["c"]} + - {owner: "redis", name: "redis", languages: ["c"]} + - {owner: "mongodb", name: "mongo", languages: ["cpp"], max_files: 2000} + + # ---------- Big data (5) ---------- + - {owner: "apache", name: "spark", languages: ["scala"], max_files: 2000} + - {owner: "apache", name: "flink", languages: ["java"], max_files: 2000} + - {owner: "apache", name: "beam", languages: ["java", "python"]} + - {owner: "apache", name: "airflow", languages: ["python"], max_files: 2000} + - {owner: "airbnb", name: "airflow", languages: ["python"]} + + # ---------- Data engineering (3) ---------- + - {owner: "dbt-labs", name: "dbt-core", languages: ["python"]} + - {owner: "pallets", name: "jinja", languages: ["python"]} + - {owner: "great-expectations", name: "great_expectations", languages: ["python"]} + + # ---------- Algorithms / data structures (5) ---------- + - {owner: "TheAlgorithms", name: "Python", languages: ["python"], max_files: 2000} + - {owner: "TheAlgorithms", name: "C", languages: ["c"]} + - {owner: "TheAlgorithms", name: "Java", languages: ["java"]} + - {owner: "TheAlgorithms", name: "Go", languages: ["go"]} + - {owner: "keon", name: "algorithms", languages: ["python"]} + + # ---------- Compilers / Languages (3) ---------- + - {owner: "rust-lang", name: "chalk", languages: ["rust"]} + - {owner: "tree-sitter", name: "tree-sitter", languages: ["c", "rust"]} + - {owner: "vlang", name: "v", languages: ["v"]} + + # ---------- Editors / IDEs (3) ---------- + - {owner: "microsoft", name: "vscode", languages: ["typescript"], max_files: 3000} + - {owner: "neovim", name: "neovim", languages: ["c", "lua"], max_files: 2000} + - {owner: "emacs", name: "emacs", languages: ["c", "emacs-lisp"], max_files: 2000} + + # ---------- DevTools (5) ---------- + - {owner: "cli", name: "cli", languages: ["go"]} + - {owner: "junegunn", name: "fzf", languages: ["go"]} + - {owner: "tmux", name: "tmux", languages: ["c"]} + - {owner: "nvie", name: "gitflow", languages: ["shell"]} + - {owner: "nvbn", name: "thefuck", languages: ["python"]} + + # ---------- Security / Crypto (3) ---------- + - {owner: "pyca", name: "pyopenssl", languages: ["python"]} + - {owner: "openssl", name: "openssl", languages: ["c"], max_files: 2000} + - {owner: "libressl-portable", name: "openbsd", languages: ["c"]} + + # ---------- Blockchain / Web3 (5) ---------- + - {owner: "ethereum", name: "go-ethereum", languages: ["go"], max_files: 2000} + - {owner: "bitcoin", name: "bitcoin", languages: ["cpp"], max_files: 2000} + - {owner: "solana-labs", name: "solana", languages: ["rust"], max_files: 2000} + - {owner: "OpenZeppelin", name: "openzeppelin-contracts", languages: ["solidity"]} + - {owner: "chainlink", name: "contracts", languages: ["solidity"]} + + # ---------- Vietnamese-specific (5) ---------- + - {owner: "Vietnamese-data-science", name: "vdsc", languages: ["python"]} + - {owner: "vinbigdata-medical", name: "vinbigdata", languages: ["python"]} + - {owner: "undertheseanlp", name: "underthesea", languages: ["python"]} + - {owner: "vietai", name: "vietai-website", languages: ["python"]} + - {owner: "vncorenlp", name: "VnCoreNLP", languages: ["java"]} + + # ---------- Open source sample projects (10) ---------- + - {owner: "httpie", name: "httpie", languages: ["python"]} + - {owner: "ansible", name: "awx", languages: ["python"]} + - {owner: "zulip", name: "zulip", languages: ["python"], max_files: 2000} + - {owner: "mailpile", name: "Mailpile", languages: ["python"]} + - {owner: "satwikkansal", name: "wtfpython", languages: ["python"]} + - {owner: "karpathy", name: "nanoGPT", languages: ["python"]} + - {owner: "karpathy", name: "micrograd", languages: ["python"]} + - {owner: "milesmcc", name: "shamir-secret-sharing", languages: ["python"]} + - {owner: "madewithml", name: "basics", languages: ["python"]} + - {owner: "GokuAI", name: "alpaca-lora", languages: ["python"]} + +# ============================================================================= +# HuggingFace datasets — curated (~50) +# ============================================================================= +huggingface: + enabled: true + cache_dir: "./data_cache/hf" + + datasets: + # ---------- Code datasets (15) ---------- + - {name: "codeparrot/codeparrot-clean", max_samples: 100000, language: "python"} + - {name: "codeparrot/github-code", max_samples: 50000, language: "multiple"} + - {name: "bigcode/the-stack-dedup", max_samples: 50000, language: "multiple"} + - {name: "bigcode/the-stack-v2-train-full-ids", max_samples: 20000} + - {name: "bigcode/starcoder2data", max_samples: 30000} + - {name: "nampdn-ai/tiny-codes", max_samples: 50000, language: "multiple"} + - {name: "HuggingFaceH4/CodeAlpaca_20K", max_samples: 20000, language: "python"} + - {name: "sahil2801/codealpaca", max_samples: 10000} + - {name: "nickroany/Evol-Instruct-Code", max_samples: 10000} + - {name: "iamtarun/codecontest", max_samples: 5000} + - {name: "openai/human-eval", max_samples: 1000} + - {name: "google-research-datasets/mbpp", max_samples: 1000} + - {name: "KaravanG/bqc-leaderboard", max_samples: 5000} + - {name: "bigcode/commitpackft", max_samples: 10000} + - {name: "bigcode/self-oss-instruct", max_samples: 10000} + + # ---------- General text / web (15) ---------- + - {name: "wikimedia/wikipedia", subset: "20231101.vi", max_samples: 50000} + - {name: "wikimedia/wikipedia", subset: "20231101.en", max_samples: 50000} + - {name: "oscar-corpus/OSCAR-2301", subset: "vi", max_samples: 30000} + - {name: "oscar-corpus/OSCAR-2301", subset: "en", max_samples: 30000} + - {name: "c4", subset: "en", max_samples: 50000} + - {name: "c4", subset: "vi", max_samples: 20000} + - {name: "allenai/dolma", max_samples: 50000} + - {name: "EleutherAI/pile", max_samples: 30000} + - {name: "HuggingFaceFW/fineweb", subset: "sample-10BT", max_samples: 50000} + - {name: "HuggingFaceFW/fineweb-edu", max_samples: 30000} + - {name: "allenai/peS2o", max_samples: 20000} + - {name: "allenai/dolma", subset: "v1_5-sample", max_samples: 20000} + - {name: "togethercomputer/RedPajama-Data-1T-Sample", max_samples: 20000} + - {name: "open-web-math/open-web-math", max_samples: 30000} + - {name: "math-ai/stack-math", max_samples: 20000} + + # ---------- Instruction-tuning (15) ---------- + - {name: "HuggingFaceH4/ultrachat_200k", max_samples: 50000} + - {name: "Open-Orca/OpenOrca", max_samples: 30000} + - {name: "teknium/OpenHermes-2.5", max_samples: 50000} + - {name: "databricks/databricks-dolly-15k", max_samples: 15000} + - {name: "tatsu-lab/alpaca", max_samples: 50000} + - {name: "vicgalle/configurable-system-prompts", max_samples: 10000} + - {name: "WizardLMTeam/WizardLM_evol_instruct_70k", max_samples: 30000} + - {name: "allenai/tulu-3-sft-mixture", max_samples: 50000} + - {name: "allenai/tulu-3-sft-personas-instruction-following", max_samples: 20000} + - {name: "allenai/RLVR-IFeval", max_samples: 10000} + - {name: "HuggingFaceH4/no_robots", max_samples: 10000} + - {name: "lmsys/lmsys-chat-1m", max_samples: 30000} + - {name: "sharegpt/sharegpt_vicuna_unfiltered", max_samples: 20000} + - {name: "openchat/openchat_3.5", max_samples: 10000} + - {name: "OpenAssistant/oasst1", max_samples: 30000} + + # ---------- Math (10) ---------- + - {name: "meta-math/MetaMathQA", max_samples: 50000} + - {name: "gsm8k", max_samples: 10000} + - {name: "lighteval/MATH", max_samples: 10000} + - {name: "hendrycks/competition_math", max_samples: 10000} + - {name: "math-ai/AQuA", max_samples: 5000} + - {name: "hendrycks/MATH", max_samples: 10000} + - {name: "openai/grade_school_math", max_samples: 8000} + - {name: "tasksource/strategyqa", max_samples: 5000} + - {name: "allenai/ai2_arc", max_samples: 5000} + - {name: "openai/openai_humaneval", max_samples: 1000} + + # ---------- Vietnamese-specific (10) ---------- + - {name: "vietgpt/news_corpus", max_samples: 30000} + - {name: "vietgpt/vietgpt-wiki", max_samples: 20000} + - {name: "PhoAT/PhoBERT", max_samples: 10000} + - {name: "vinbigdata/uit-viic", max_samples: 5000} + - {name: "sonlam/ Vietnamese-translation-alpaca", max_samples: 10000} + - {name: "nhoxquyxoem/vi-alpaca-vicuna-instruct", max_samples: 5000} + - {name: "VietnamAIHub/Vietnamese_translation", max_samples: 10000} + - {name: "vietnamese-data-science/vi-news", max_samples: 10000} + - {name: "duongkstn/mt-vi-train", max_samples: 5000} + - {name: "botran/vagrant-vi", max_samples: 5000} + +# ============================================================================= +# arXiv — scientific papers +# ============================================================================= +arxiv: + enabled: true + delay_seconds: 3.0 + + queries: + - "transformer architecture" + - "mixture of experts" + - "large language model" + - "attention mechanism" + - "code generation" + - "program synthesis" + - "neural machine translation" + - "retrieval augmented generation" + - "instruction tuning" + - "reinforcement learning human feedback" + - "chain of thought reasoning" + - "prompt engineering" + - "fine-tuning language model" + - "quantization neural network" + - "knowledge distillation" + - "multi-agent systems" + - "tool use language model" + - "code completion" + - "static analysis" + - "program verification" + - "diffusion models" + - "vision transformer" + - "multimodal learning" + - "federated learning" + - "differential privacy" + - "graph neural network" + - "reinforcement learning" + - "meta learning" + - "few-shot learning" + - "self-supervised learning" + - "contrastive learning" + - "long context language model" + - "RoPE" + - "flash attention" + - "sliding window attention" + - "ALiBi" + - "RLHF" + - "DPO" + - "GRPO" + - "RLAIF" + - "agent benchmark" + +# ============================================================================= +# Wikipedia — encyclopedic text +# ============================================================================= +wikipedia: + enabled: true + languages: ["vi", "en"] + topics: + vi: + - "Trí tuệ nhân tạo" + - "Học máy" + - "Mạng nơ-ron nhân tạo" + - "Python (ngôn ngữ lập trình)" + - "JavaScript" + - "Linux" + - "Cơ sở dữ liệu" + - "Thuật toán" + - "Cấu trúc dữ liệu" + - "Lập trình hướng đối tượng" + - "API" + - "JSON" + - "Git" + - "Hệ điều hành" + - "Học sâu" + - "Xử lý ngôn ngữ tự nhiên" + - "Big data" + - "Điện toán đám mây" + - "Cryptography" + - "Blockchain" + - "Microservices" + - "Docker (phần mềm)" + - "Kubernetes" + - "Terraform (phần mềm)" + - "Ansible" + - "PostgreSQL" + - "Redis" + - "MongoDB" + en: + - "Artificial intelligence" + - "Machine learning" + - "Neural network" + - "Python (programming language)" + - "JavaScript" + - "Linux" + - "Database" + - "Algorithm" + - "Data structure" + - "Object-oriented programming" + - "API" + - "JSON" + - "Git" + - "Operating system" + - "Deep learning" + - "Natural language processing" + - "Big data" + - "Cloud computing" + - "Transformer (deep learning model)" + - "Large language model" + - "Diffusion model" + - "GPT" + - "BERT" + - "Mixture of experts" + - "FlashAttention" + - "RoPE" + - "Long context language model" + +# ============================================================================= +# StackOverflow — Q&A +# ============================================================================= +stackoverflow: + enabled: true + page_size: 100 + min_score: 5 + tags: + - "python" + - "javascript" + - "java" + - "c#" + - "php" + - "android" + - "html" + - "jquery" + - "c++" + - "css" + - "ios" + - "mysql" + - "sql" + - "node.js" + - "reactjs" + - "ruby-on-rails" + - "vue.js" + - "typescript" + - "docker" + - "git" + - "go" + - "rust" + - "machine-learning" + - "deep-learning" + - "pytorch" + - "tensorflow" + - "pandas" + - "numpy" + - "regex" + - "algorithm" + - "bash" + - "shell" + - "linux" + - "kubernetes" + - "terraform" + - "ansible" + - "aws" + - "azure" + - "gcp" + - "redis" + - "elasticsearch" + - "kafka" + - "rabbitmq" + - "postgresql" + - "mongodb" + - "sqlite" + +# ============================================================================= +# v0.3 NEW SOURCES +# ============================================================================= + +# The-Stack v2 — BigCode's massive code dataset +the_stack: + enabled: true + cache_dir: "./data_cache/the_stack" + version: "v2" + # Top languages by sample count (rest skipped to keep size manageable) + languages: + - "python" + - "javascript" + - "typescript" + - "java" + - "go" + - "rust" + - "c" + - "cpp" + - "csharp" + - "ruby" + - "php" + - "swift" + - "kotlin" + - "scala" + - "shell" + - "sql" + max_samples_per_language: 5000 + min_stars: 0 # include all repos regardless of stars + license_filter: ["mit", "apache-2.0", "bsd-3-clause", "bsd-2-clause", "mpl-2.0", "unlicense"] + +# StarCoder2 training data (github-code + commits + notebooks) +starcoder2_data: + enabled: true + cache_dir: "./data_cache/starcoder2" + components: + - "github_code" # code files + - "github_commits" # commit diffs (good for editing tasks) + - "github_jupyter" # notebook cells (markdown + code) + max_samples_per_component: 20000 + languages: + - "python" + - "javascript" + - "typescript" + - "java" + - "go" + - "rust" + - "c" + - "cpp" + +# Python-Alpaca — high-quality Python instruction data +python_alpaca: + enabled: true + cache_dir: "./data_cache/python_alpaca" + sources: + - {name: "sahil2801/codealpaca", max_samples: 20000} + - {name: "HuggingFaceH4/CodeAlpaca_20K", max_samples: 20000} + - {name: "nickroany/Evol-Instruct-Code", max_samples: 15000} + - {name: "TheBloke/CodeAlpaca-13B", max_samples: 5000} + - {name: "codeparrot/codeparrot-clean", max_samples: 50000} + - {name: "nampdn-ai/tiny-codes", max_samples: 50000} + +# Kaggle — competition kernels & datasets metadata +kaggle: + enabled: false # disabled by default — requires API key + cache_dir: "./data_cache/kaggle" + api_key_env: "KAGGLE_API_KEY" + competitions: + - "titanic" + - "house-prices-advanced-regression-techniques" + - "digit-recognizer" + - " Spaceship-Titanic" + - "favorita-grocery-sales-forecasting" + max_kernels_per_competition: 100 + +# ============================================================================= +# Processing pipeline settings +# ============================================================================= +processing: + cleaner: + remove_html: true + remove_urls: false + normalize_whitespace: true + min_length: 50 + max_length: 100000 + + quality_filter: + min_length: 50 + max_length: 100000 + min_words: 10 + min_unique_ratio: 0.3 + max_repetition: 0.5 + + deduplicator: + ngram_size: 5 + num_perm: 128 + similarity_threshold: 0.8 + + # v0.3 NEW processors + language_id: + enabled: true + # Identify language of each text sample (drops mislabelled) + min_confidence: 0.85 + allowed_languages: ["vi", "en", "code"] + + code_quality: + enabled: true + # Score code samples (1-10), drop samples below threshold + min_score: 6.0 + factors: + has_docstring: 1.5 + has_type_hints: 1.0 + no_print: 0.5 + no_eval: 1.0 + no_bare_except: 1.0 + reasonable_length: 1.0 # 10-500 lines + has_test: 2.0 # bonus for adjacent test file + + curriculum: + stages: + - {name: "easy", min_length: 50, max_length: 500, min_quality: 0.7} + - {name: "medium", min_length: 500, max_length: 5000, min_quality: 0.6} + - {name: "hard", min_length: 5000, max_length: 30000, min_quality: 0.7} + - {name: "expert", min_length: 30000, max_length: 100000, min_quality: 0.8} + +# ============================================================================= +# Token budget estimation (v0.3 NEW) +# ============================================================================= +token_budget: + total_target_tokens: 500_000_000_000 # 500B tokens (for 30B model pretrain) + distribution: + code: 0.40 # 40% code (The-Stack, StarCoder2-data, GitHub) + text: 0.30 # 30% natural text (Wikipedia, C4, OSCAR) + instruction: 0.15 # 15% instruction-tuning (Alpaca, ShareGPT) + math: 0.10 # 10% math (GSM8K, MATH, MetaMathQA) + vietnamese: 0.05 # 5% Vietnamese-specific diff --git a/data/README.md b/data/README.md new file mode 100644 index 0000000000000000000000000000000000000000..a0fc2ea6c81e2d58d30126255443e2262321ada1 --- /dev/null +++ b/data/README.md @@ -0,0 +1,9 @@ +# Data directory + +Thư mục này chứa dữ liệu huấn luyện (nếu có). + +This directory contains training data (if any). + +Hiện tại, training data được hardcoded trong `nexus/training/dataset.py`. + +Currently, training data is hardcoded in `nexus/training/dataset.py`. diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md new file mode 100644 index 0000000000000000000000000000000000000000..f5062f769f7265707fd929c37d08ab8a28c0b850 --- /dev/null +++ b/docs/ARCHITECTURE.md @@ -0,0 +1,95 @@ +# Kiến trúc Nexus Coder / Nexus Coder Architecture + +## Tổng quan / Overview + +Nexus Coder v0.1 sử dụng kiến trúc **Mixture of Experts (MoE) Transformer** tương tự Mixtral 8x7B và DeepSeek-V3. + +Nexus Coder v0.1 uses a **Mixture of Experts (MoE) Transformer** architecture similar to Mixtral 8x7B and DeepSeek-V3. + +## Các thành phần / Components + +### 1. Token Embedding +- Vocab size: 32,000 +- Hidden size: 2,048 +- Tokens được nhúng thành vector 2048 chiều + +### 2. Grouped Query Attention (GQA) +- 16 query heads +- 4 KV heads (ratio 4:1) +- Head dimension: 128 +- Giảm 4x memory cho KV cache so với MHA truyền thống + +### 3. Rotary Position Embedding (RoPE) +- Base: 10,000 +- Hỗ trợ tối đa 50,000 positions +- Cho phép model hiểu vị trí tương đối giữa các tokens + +### 4. RMSNorm +- Thay thế LayerNorm truyền thống +- Không có bias, không trừ mean +- Nhanh hơn ~10-20% + +### 5. SwiGLU Activation +- `SiLU(gate(x)) * up(x)` +- Hiệu quả hơn ReLU/GELU +- Có 3 ma trận: gate, up, down (3 * hidden * intermediate params) + +### 6. Mixture of Experts (MoE) - Cốt lõi +- **24 experts** tổng cộng (mỗi expert là một SwiGLU FFN) +- **3 active experts** mỗi token (top-3 routing) +- Router: linear layer (hidden_size → num_experts) +- Load balancing loss: auxiliary loss để tránh expert collapse + +#### Routing Algorithm +``` +1. Router tính gate_logits = W_router @ x +2. routing_weights = softmax(gate_logits) +3. top_k_weights, top_k_indices = topk(routing_weights, k=3) +4. Normalize top_k_weights +5. Mỗi token đi qua 3 expert được chọn +6. Output = sum(weight_i * expert_i(x)) +``` + +## Tính toán tham số / Parameter Math + +``` +Embedding: vocab_size × hidden = 32000 × 2048 = 65.5M +Per layer attn: 2048² + 2×(2048×512) + 2048² = 10.5M (Q, K, V, O with GQA) +Per expert: 3 × 2048 × 5632 = 34.6M (gate + up + down) +Per layer MoE: 24 × 34.6M = 830M (total) + 3 × 34.6M = 104M (active) +Per layer total: 10.5M + 830M = 840.5M +12 layers: 10,086M +LM head: 65.5M +──────────────────────────────────── +TOTAL: 10,223M ≈ 10.22B ✓ +ACTIVE: 65.5 + 12×(10.5 + 104) + 65.5 = 1,503M ≈ 1.50B ✓ +``` + +## Workflow + +### Training Workflow +1. Tokenize input text → token IDs +2. Embed tokens → hidden states [B, L, H] +3. For each layer: + - Pre-norm → Attention → residual + - Pre-norm → MoE (router + experts) → residual +4. Final norm → LM head → logits +5. Compute cross-entropy loss + aux loss +6. Backpropagation + +### Inference Workflow +1. Tokenize prompt +2. Forward pass through all layers +3. Get logits for last position +4. Apply temperature, top-k, top-p +5. Sample next token +6. Append to sequence, repeat + +## Tối ưu / Optimizations + +- **KV Cache**: Cache K, V từ các token trước để tăng tốc generation +- **GQA**: Giảm memory và computation cho attention +- **Pre-norm**: Ổn định hơn post-norm trong training +- **Mixed Precision**: Hỗ trợ fp16/bf16 để tiết kiệm memory +- **Gradient Checkpointing**: Đánh đổi compute lấy memory (chưa implement trong v0.1) diff --git a/docs/DATA.md b/docs/DATA.md new file mode 100644 index 0000000000000000000000000000000000000000..2e6fc20cfcc04301dcc5f93a52bcc2fa99d5402f --- /dev/null +++ b/docs/DATA.md @@ -0,0 +1,188 @@ +# Data Pipeline Documentation + +Nexus Coder v0.2 có pipeline thu thập và xử lý training data hoàn chỉnh. + +## Overview + +``` +┌─────────────┐ ┌──────────────┐ ┌─────────────┐ ┌──────────────┐ +│ COLLECT │ ──> │ PROCESS │ ──> │ TRAIN │ ──> │ EVALUATE │ +│ (5 sources) │ │ (4 stages) │ │ (curriculum)│ │ (8 benches) │ +└─────────────┘ └──────────────┘ └─────────────┘ └──────────────┘ +``` + +## Sources (Collectors) + +### 1. GitHub +- **60+ curated repos** (Python, JS, TS, Go, Rust, C, C++) +- Categories: Python core, Data science, ML/DL, Web, CLI, Async, Database, Tools +- Quality filter: size, content, auto-generated detection +- File extensions: .py, .js, .ts, .go, .rs, .java, .c, .cpp, .sql, .sh, .md + +### 2. HuggingFace +- **20+ curated datasets**: + - Code: codeparrot, the-stack, CodeAlpaca + - Text: Wikipedia (vi, en), C4, OSCAR + - Chat: UltraChat, OpenOrca, OpenHermes, Dolly + - Math: MetaMathQA, GSM8K, MATH + - Vietnamese: news_corpus, PhoATC + +### 3. arXiv +- 20 curated queries (transformer, MoE, LLM, code generation, etc.) +- Categories: cs.CL, cs.LG, cs.AI, cs.SE, cs.PL, cs.CV, stat.ML +- Rate limit: 1 request per 3 seconds + +### 4. Wikipedia +- Vietnamese + English +- 20 curated topics per language +- Random article collection supported + +### 5. StackOverflow +- 30 curated tags (python, javascript, java, etc.) +- Filter by minimum score (default: 5) +- Includes accepted answers +- Rate limit: 30 req/s + +## Processing Pipeline + +### Stage 1: Clean (TextCleaner) +- HTML tag removal +- Unicode normalization (NFC) +- Control character removal +- HTML entity decoding +- Whitespace normalization +- Encoding fix + +### Stage 2: Format (CodeFormatter) +- Language detection (by extension + patterns) +- Trailing whitespace removal +- Excessive blank line removal (max 2 consecutive) +- Leading/trailing blank line removal +- Markdown fence wrapping + +### Stage 3: Quality Filter (QualityFilter) +- Length check (50-100,000 chars) +- Word count (min 10) +- Unique word ratio (min 0.3) +- Repetition score (max 0.5) +- Spam pattern detection +- Code presence bonus + +### Stage 4: Deduplicate (Deduplicator) +- Exact hash dedup (MD5) +- MinHash LSH for near-duplicates +- 128 permutations, 5-gram +- Jaccard threshold: 0.8 + +## Curriculum Learning + +4-stage curriculum: + +| Stage | Difficulty | Length | Quality | Description | +|-------|-----------|--------|---------|-------------| +| 1 | EASY | 50-500 | ≥0.7 | Short basic text - vocabulary | +| 2 | MEDIUM | 500-5000 | ≥0.6 | Standard length - grammar | +| 3 | HARD | 5000-30000 | ≥0.7 | Long technical - deep understanding | +| 4 | EXPERT | 30000-100000 | ≥0.8 | Multi-step reasoning | + +## Usage + +### Collect raw data + +```bash +# Collect from all sources +python scripts/collect_data.py --source all --output ./data/raw + +# Or specific source +python scripts/collect_data.py --source github --max-repos 10 +python scripts/collect_data.py --source huggingface --max-datasets 5 +``` + +### Process raw data + +```bash +python scripts/prepare_dataset.py --input ./data/raw --output ./data/processed +``` + +### Train with external data + +```bash +python scripts/train.py --config large --include-external --steps 5000 +``` + +## Output Format + +Processed data saved as JSONL files by difficulty: + +``` +data/processed/ +├── train_easy.jsonl # Stage 1 samples +├── train_medium.jsonl # Stage 2 samples +├── train_hard.jsonl # Stage 3 samples +├── train_expert.jsonl # Stage 4 samples +└── processing_stats.json # Statistics +``` + +Each JSONL line: +```json +{ + "text": "...", + "source": "github:python/cpython", + "language": "python", + "metadata": { + "file_path": "Lib/os.py", + "size": 45678, + "quality_score": 0.85, + "quality": {"score": 0.85, "length": 45678, "word_count": 1200, "has_code": true}, + "cleaned": true, + "cleaned_length": 45678, + "formatted": true, + "detected_language": "python" + } +} +``` + +## Environment Variables + +```bash +# GitHub API (for search) +export GITHUB_TOKEN=ghp_xxx + +# HuggingFace Hub (for gated datasets) +export HF_TOKEN=hf_xxx + +# Web search API (optional) +export SEARCH_API_KEY=xxx +export BRAVE_SEARCH_API_KEY=xxx +``` + +## Estimate Data Volume + +| Source | Estimated samples | Estimated size | +|--------|------------------|----------------| +| GitHub (60 repos) | ~50,000 files | ~500 MB | +| HuggingFace (20 datasets) | ~200,000 samples | ~2 GB (streamed) | +| arXiv (20 queries) | ~400 papers | ~50 MB | +| Wikipedia (vi+en) | ~40 articles | ~5 MB | +| StackOverflow (30 tags) | ~1,500 Q&A | ~10 MB | +| **Total** | **~250,000 samples** | **~2.5 GB** | + +After deduplication and quality filter: ~150,000 high-quality samples. + +## Custom Sources + +Add your own collector: + +```python +from nexus.data.collectors.base import Collector + +class MyCollector(Collector): + def collect(self): + # Yield samples as dicts + yield { + "text": "...", + "source": "my_source", + "language": "en", + "metadata": {...}, + } +``` diff --git a/docs/SKILLS.md b/docs/SKILLS.md new file mode 100644 index 0000000000000000000000000000000000000000..ab80a22eb42df38be439154a98f960cdb13c7ff7 --- /dev/null +++ b/docs/SKILLS.md @@ -0,0 +1,175 @@ +# Skills Documentation + +Nexus Coder v0.2 có 15 skills chuyên môn, được tổ chức theo 5 categories. + +## Categories + +| Category | Skills | +|----------|--------| +| CODE | code_generation, code_review, code_refactor, debugging, documentation, testing | +| REASONING | algorithm_design, reasoning, math_skill | +| LANGUAGE | translation, summarization | +| DATA | data_analysis, sql_generation | +| SECURITY | security_audit | +| DEVOPS | performance_optimization | + +## Skill List + +### 1. code_generation +- **Category**: CODE +- **Priority**: HIGH +- **Description**: Sinh code từ mô tả tự nhiên +- **Languages**: Python, JavaScript, TypeScript, Go, Rust, C++, Java, SQL +- **Example**: "Viết hàm Python tính fibonacci" + +### 2. code_review +- **Category**: CODE +- **Priority**: HIGH +- **Description**: Review code toàn diện +- **Checks**: bugs, security, performance, style, error handling, type safety +- **Example**: "Review đoạn code này giúp tôi" + +### 3. code_refactor +- **Category**: CODE +- **Priority**: MEDIUM +- **Description**: Refactor code an toàn +- **Patterns**: Extract Method/Class, Rename, Move, Replace Conditional, etc. +- **Example**: "Refactor hàm này cho clean hơn" + +### 4. debugging +- **Category**: CODE +- **Priority**: CRITICAL +- **Description**: Debug code với 7-step protocol +- **Supports**: Python, JavaScript, Java, Go, Rust, C++, Ruby +- **Example**: "Fix lỗi IndexError trong hàm này" + +### 5. documentation +- **Category**: CODE +- **Priority**: MEDIUM +- **Description**: Sinh tài liệu tự động +- **Types**: Docstrings (Google/NumPy/Sphinx), README, API ref, tutorials +- **Example**: "Sinh docstring cho hàm này" + +### 6. testing +- **Category**: CODE +- **Priority**: HIGH +- **Description**: Sinh tests +- **Types**: unit, integration, E2E, property-based, mutation, fuzz, snapshot +- **Frameworks**: pytest, unittest, jest, vitest, mocha, cargo test, JUnit +- **Example**: "Viết unit tests cho class User" + +### 7. algorithm_design +- **Category**: REASONING +- **Priority**: MEDIUM +- **Description**: Thiết kế thuật toán +- **Approaches**: Brute force, Greedy, D&C, DP, Backtracking, Graph algorithms +- **Example**: "Tối ưu thuật toán này từ O(n²) xuống O(n log n)" + +### 8. data_analysis +- **Category**: DATA +- **Priority**: MEDIUM +- **Description**: Phân tích dữ liệu +- **Steps**: Loading, cleaning, statistics, correlation, outliers, visualization +- **Libraries**: pandas, numpy, scipy, matplotlib, seaborn, plotly +- **Example**: "Phân tích dataset này và tìm insights" + +### 9. translation +- **Category**: LANGUAGE +- **Priority**: MEDIUM +- **Description**: Dịch song ngữ Việt-Anh +- **Pairs**: vi↔en, vi↔zh, vi↔ja, vi↔ko, vi↔fr +- **Example**: "Dịch đoạn văn này sang tiếng Anh" + +### 10. summarization +- **Category**: LANGUAGE +- **Priority**: MEDIUM +- **Description**: Tóm tắt văn bản +- **Methods**: extractive, abstractive, key phrase, topic modeling +- **Example**: "Tóm tắt bài viết này trong 3 câu" + +### 11. reasoning +- **Category**: REASONING +- **Priority**: HIGH +- **Description**: Suy luận đa bước +- **Strategies**: CoT, ToT, Self-Consistency, Reflexion, ReAct, Least-to-Most +- **Example**: "Tại sao bầu trời màu xanh?" + +### 12. math_skill +- **Category**: REASONING +- **Priority**: HIGH +- **Description**: Giải toán đa cấp +- **Domains**: arithmetic, algebra, calculus, linear algebra, probability, statistics +- **Tools**: sympy, numpy, scipy +- **Example**: "Tính đạo hàm của x³ + 2x²" + +### 13. sql_generation +- **Category**: DATA +- **Priority**: HIGH +- **Description**: Sinh SQL queries +- **Dialects**: PostgreSQL, MySQL, SQLite, SQL Server, Oracle, BigQuery, Snowflake +- **Example**: "Viết SQL tìm top 10 khách hàng" + +### 14. security_audit +- **Category**: SECURITY +- **Priority**: CRITICAL +- **Description**: Audit bảo mật +- **Standards**: OWASP Top 10, SAST, dependency vulnerabilities +- **Tools**: bandit, semgrep, safety, pip-audit, trufflehog +- **Example**: "Audit code này cho security issues" + +### 15. performance_optimization +- **Category**: DEVOPS +- **Priority**: MEDIUM +- **Description**: Tối ưu hiệu năng +- **Categories**: algorithmic, memory, concurrency, caching, I/O, Python-specific +- **Tools**: cProfile, line_profiler, memory_profiler, py-spy +- **Example**: "Tối ưu hàm này đang chạy chậm" + +## Usage + +```python +from nexus.skills import get_global_registry +from nexus.skills.base import SkillContext + +registry = get_global_registry() + +# List all skills +print(registry.list_skills()) + +# Route prompt to best skill +skill = registry.route("Viết hàm Python tính giai thừa") +print(f"Selected: {skill.name}") + +# Execute skill +context = SkillContext(prompt="Viết hàm Python tính giai thừa") +result = skill.execute(context) +print(result.output) +``` + +## Custom Skills + +Tạo skill tùy chỉnh: + +```python +from nexus.skills.base import Skill, SkillResult, SkillContext, SkillCategory, SkillPriority + +class MyCustomSkill(Skill): + category = SkillCategory.CODE + priority = SkillPriority.MEDIUM + keywords = ["custom", "riêng"] + + @property + def name(self) -> str: + return "my_custom_skill" + + @property + def description(self) -> str: + return "My custom skill description" + + def execute(self, context: SkillContext) -> SkillResult: + return SkillResult(success=True, output="Custom result") + +# Register +from nexus.skills import get_global_registry +get_global_registry().register(MyCustomSkill()) +``` diff --git a/docs/TOOLS.md b/docs/TOOLS.md new file mode 100644 index 0000000000000000000000000000000000000000..5285dd2d536bead95da4e5bd456fc27007c78893 --- /dev/null +++ b/docs/TOOLS.md @@ -0,0 +1,162 @@ +# Tools Documentation + +Nexus Coder v0.2 có 18+ tools để tương tác với môi trường. + +## Safety Levels + +| Level | Icon | Description | +|-------|------|-------------| +| SAFE | ✓ | Read-only, no side effects | +| MODERATE | ⚠ | Writes to local files | +| DANGEROUS | ⚡ | Executes commands, network ops | +| DESTRUCTIVE | 💀 | Can delete data, requires confirmation | + +## Tools by Category + +### FILE Operations +- `file_read` (✓) - Đọc file text +- `file_write` (⚠) - Ghi file (overwrite/append) +- `file_list` (✓) - Liệt kê files với glob +- `file_delete` (💀) - Xóa file/thư mục + +### EXEC +- `shell_exec` (⚡) - Execute bash commands +- `python_exec` (⚡) - Execute Python code (sandboxed) +- `git_ops` (⚡) - Git commands + +### WEB +- `http_request` (⚠) - HTTP GET/POST/PUT/DELETE +- `web_fetch` (✓) - Fetch webpage, extract text +- `web_search` (✓) - Search web + +### CODE +- `code_search` (✓) - Regex search trong code +- `code_lint` (✓) - Lint code (ruff, flake8, pylint) +- `code_format` (⚠) - Format code (black, autopep8, isort) +- `regex_search` (✓) - Regex search trong files + +### MATH +- `calculator` (✓) - Safe math expression eval + +### PARSER +- `json_parse` (✓) - Parse JSON với query support +- `yaml_parse` (✓) - Parse YAML +- `csv_parse` (✓) - Parse CSV + +### SYSTEM +- `datetime` (✓) - DateTime operations + timezone + +### NETWORK +- `dns_lookup` (✓) - DNS lookup (A, AAAA, MX, NS, CNAME, TXT) +- `ping` (✓) - Ping host + +### CRYPTO +- `hash` (✓) - Compute hash (md5, sha1, sha256, sha512, blake2) +- `encrypt` (⚡) - AES-256-GCM encrypt/decrypt + +### FILE (Archive) +- `archive` (⚠) - ZIP/TAR create/extract/list + +## Usage + +```python +from nexus.tools import get_global_registry, ToolContext + +registry = get_global_registry() + +# List all tools +print(registry.list_tools()) + +# Execute tool +from nexus.tools.base import ToolContext +ctx = ToolContext(working_dir="/tmp") +result = registry.execute("file_read", {"path": "/etc/hostname"}, ctx) +print(result.output) + +# Check safety +tool = registry.get("file_delete") +print(f"Safety: {tool.safety.value}") +``` + +## Audit Log + +All tool calls are logged to `./logs/tool_audit.jsonl`: + +```json +{ + "timestamp": 1234567890.123, + "tool": "file_write", + "safety": "moderate", + "args": {"path": "/tmp/test.txt", "content": "hello"}, + "working_dir": ".", + "user_id": null, + "success": true, + "return_code": 0, + "duration": 0.001 +} +``` + +## Safety Features + +1. **Confirmation required** for DANGEROUS and DESTRUCTIVE tools +2. **Dry-run mode** to preview actions without executing +3. **Pre-hooks** for rate limiting, auth checks +4. **Post-hooks** for metrics, notifications +5. **Audit log** for compliance +6. **Blocked commands** for known dangerous patterns + +## Custom Tools + +```python +from nexus.tools.base import Tool, ToolResult, ToolContext, ToolCategory, ToolSafety + +class MyTool(Tool): + category = ToolCategory.FILE + safety = ToolSafety.SAFE + + @property + def name(self) -> str: + return "my_tool" + + @property + def description(self) -> str: + return "My custom tool" + + @property + def parameters(self) -> dict: + return { + "type": "object", + "properties": {"input": {"type": "string"}}, + "required": ["input"], + } + + def execute(self, args, context): + return ToolResult( + success=True, + output=f"Processed: {args['input']}", + ) + +# Register +from nexus.tools import get_global_registry +get_global_registry().register(MyTool()) +``` + +## Tool Calling via Natural Language + +Agent có thể detect tool calls từ natural language: + +- "read file /etc/hostname" → `file_read` +- "run ls -la" → `shell_exec` +- "search for TODO in src/" → `regex_search` +- "fetch https://example.com" → `web_fetch` +- "git status" → `git_ops` + +Hoặc JSON format: +```json +{"tool": "file_read", "args": {"path": "/etc/hostname"}} +``` + +Or @-mention: +``` +@file_read path=/etc/hostname +``` diff --git a/docs/TRAINING.md b/docs/TRAINING.md new file mode 100644 index 0000000000000000000000000000000000000000..2df51a29d166f5609bb8868c93977f2402a7551c --- /dev/null +++ b/docs/TRAINING.md @@ -0,0 +1,77 @@ +# Huấn luyện Nexus Coder / Training Nexus Coder + +## Tổng quan / Overview + +Nexus Coder v0.1 có thể được huấn luyện với script `scripts/train.py`. Training data được "hardcoded" với thông tin tác giả. + +## Training Data + +Dữ liệu huấn luyện nằm trong `nexus/training/dataset.py` và chứa: +- Q&A về tác giả (Hieu Louis) +- Sample code snippets +- Small talk examples +- Cả tiếng Việt và tiếng Anh + +Để thêm dữ liệu, chỉnh sửa `AUTHOR_TRAINING_DATA` trong file đó. + +## Cấu hình / Configuration + +### Tiny config (CPU) +```bash +python scripts/train.py --steps 100 --batch_size 2 --max_length 64 +``` + +### Full 10B config (cần GPU) +```bash +python scripts/train.py --full --steps 5000 --batch_size 4 +``` + +## Yêu cầu hệ thống / System Requirements + +### Tiny config +- CPU: bất kỳ +- RAM: 2GB+ +- Disk: 100MB + +### Full 10B config +- GPU: cần nhiều GPU (VD: 4x A100 80GB) +- RAM: 64GB+ +- Disk: 50GB+ cho checkpoints +- Training time: nhiều ngày/tuần + +## Hyperparameters mặc định + +| Tham số | Giá trị | +|---------|---------| +| Learning rate | 5e-4 | +| Weight decay | 0.01 | +| Warmup steps | 100 | +| Max steps | 5000 | +| Batch size | 4 | +| Gradient accumulation | 4 | +| Save steps | 500 | +| Max grad norm | 1.0 | +| LR schedule | Cosine | +| Optimizer | AdamW (β1=0.9, β2=0.95) | + +## Outputs + +Training sẽ tạo: +- `checkpoints/nexus_coder-step-{N}.pt` - checkpoint +- `checkpoints/nexus_coder-final.pt` - final checkpoint +- `checkpoints/tokenizer.json` - trained tokenizer +- `checkpoints/training_log.json` - training log + +## Tiếp tục từ checkpoint + +```bash +# Đang cập nhật trong v0.2 +``` + +## Lưu ý / Notes + +⚠️ **v0.1 chỉ là foundation**: +- Tiny config chỉ dùng để verify code chạy được +- Full 10B cần GPU nhiều VRAM và nhiều thời gian +- Model chưa được pre-trained trên corpus lớn +- Để model trả lời thực sự, cần train thêm nhiều dữ liệu diff --git a/nexus/__init__.py b/nexus/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..1791465b2f1f13f8fe7e1b3f64c02ab13f7feac4 --- /dev/null +++ b/nexus/__init__.py @@ -0,0 +1,81 @@ +""" +Nexus Coder - Super CyberGym AI +================================ +v0.4.0 - CyberForge edition + +Model AI được tạo bởi Hieu Louis (2026) + +Tổng tham số: 423 tỷ (423B) +Tham số kích hoạt: 39 tỷ (39B active) +Cửa sổ ngữ cảnh: 3,000,000 tokens (3M) + +Kiến trúc: CyberForge MoE Transformer + - GQA + RoPE (YaRN-scaled) + RMSNorm + SwiGLU + FlashAttention-2 + - Sliding Window Attention + QK-norm + KV cache quantization + - MLP-parallel + Gradient checkpointing + - CyberGym training: Mutation Pressure + Code Genome + Expert Speciation + CEP + +Skills: 60+ · Tools: 80+ · Data sources: 8+ · Code corpus: 3000+ repos + +Tác giả: Hieu Louis +GitHub: mhieuhonda +Năm: 2026 +""" + +__version__ = "0.4.0" +__author__ = "Hieu Louis" +__github__ = "mhieuhonda" +__year__ = "2026" +__license__ = "NexusCoder Attribution License v1.0" + +# Thông tin tác giả được "huấn luyện cứng" vào model +AUTHOR_INFO = { + "name": "Hieu Louis", + "github": "mhieuhonda", + "year": "2026", + "description": ( + "Nexus Coder là dự án AI cá nhân do Hieu Louis tự xây dựng từ đầu " + "với kiến trúc CyberForge MoE tiên tiến, kết hợp CyberGym training." + ), + "model_name": "Nexus Coder", + "agent_name": "Nexus", + "version": "0.4.0", + "architecture": ( + "CyberForge MoE Transformer (GQA + RoPE/YaRN + RMSNorm + SwiGLU + " + "FlashAttention-2 + Sliding Window + QK-norm + KV-cache quant + " + "MLP-parallel + Gradient checkpointing)" + ), + "total_params": "~423B (variants: 5M tiny → 423B)", + "active_params": "~39B (variants: 2M tiny → 39B)", + "context_window": "3,000,000 tokens (3M, via YaRN + CEP)", + "python_version": "3.12.13", + "skills_count": "60+", + "tools_count": "80+", + "data_sources": "8+ (GitHub 3000+ repos, HuggingFace, arXiv, Wikipedia, StackOverflow, The-Stack, StarCoder2-data, Python-Alpaca)", + "training_methodology": "CyberForge (Mutation Pressure Training + Code Genome Init + Expert Speciation + Context Expansion Protocol)", + "training_frameworks_referenced": "litgpt, LlamaFactory, axolotl, OpenHands, omp-gym", +} + + +# Lazy import để giảm startup time +def __getattr__(name: str): + if name == "NexusConfig": + from .config import NexusConfig + return NexusConfig + if name == "NEXUS_CODER_10B_CONFIG": + from .config import NEXUS_CODER_10B_CONFIG + return NEXUS_CODER_10B_CONFIG + if name == "NEXUS_CODER_423B_CONFIG": + from .config import NEXUS_CODER_423B_CONFIG + return NEXUS_CODER_423B_CONFIG + raise AttributeError(f"module 'nexus' has no attribute {name!r}") + + +__all__ = [ + "AUTHOR_INFO", + "__version__", + "__author__", + "__github__", + "__year__", + "__license__", +] diff --git a/nexus/agent/__init__.py b/nexus/agent/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..e2d318eb3fb5c420b39e666c7576449dc78cdc9d --- /dev/null +++ b/nexus/agent/__init__.py @@ -0,0 +1,4 @@ +"""Agent package.""" +from .agent import NexusAgent + +__all__ = ["NexusAgent"] diff --git a/nexus/agent/agent.py b/nexus/agent/agent.py new file mode 100644 index 0000000000000000000000000000000000000000..a08203cac02e6d1a123d8303ec89010840872f3c --- /dev/null +++ b/nexus/agent/agent.py @@ -0,0 +1,364 @@ +""" +Nexus Agent v0.2 - AI Agent với Skills + Tools +=============================================== +Major upgrade từ v0.1: +- Tích hợp SkillRegistry (15+ skills) +- Tích hợp ToolRegistry (15+ tools) +- Memory system +- Planner cho multi-step tasks +- Tool routing thông minh +- Safety guardrails +- Audit logging +""" +from __future__ import annotations + +from typing import Optional, List, Dict, Any, Callable +import json +import os +from datetime import datetime +from pathlib import Path + +from ..config import NexusConfig +from ..inference.generator import NexusGenerator, DEFAULT_SYSTEM_PROMPT +from ..tokenizer.tokenizer import NexusTokenizer +from ..model.nexus_coder import NexusCoderForCausalLM +from .. import AUTHOR_INFO +from ..skills import SkillRegistry, get_global_registry as get_skill_registry +from ..tools import ToolRegistry, ToolContext, get_global_registry as get_tool_registry +from ..safety import SafetyFilter, get_default_guardrails +from .memory import ConversationMemory +from .planner import TaskPlanner +from .router import ToolRouter + + +class NexusAgent: + """AI Agent v0.2 - Wrapper cấp cao cho Nexus Coder. + + Features: + - Skill-based routing (15+ skills) + - Tool use (15+ tools) + - Conversation memory + - Task planning + - Safety guardrails + - Audit logging + + Usage: + agent = NexusAgent() + agent.chat() # Interactive + # or + response = agent.respond("Viết hàm fibonacci") + """ + + def __init__( + self, + generator: Optional[NexusGenerator] = None, + config: Optional[NexusConfig] = None, + name: str = "Nexus", + personality: str = "humorous", + language: str = "bilingual", + enable_logging: bool = True, + log_dir: str = "./logs", + enable_skills: bool = True, + enable_tools: bool = True, + enable_memory: bool = True, + enable_planner: bool = True, + enable_safety: bool = True, + working_dir: str = ".", + ): + self.config = config or NexusConfig() + self.name = name + self.personality = personality + self.language = language + self.author_info = AUTHOR_INFO + self.working_dir = working_dir + + # Generator (model + tokenizer) + if generator is None: + self.generator = NexusGenerator( + model=NexusCoderForCausalLM(self.config), + tokenizer=NexusTokenizer(), + config=self.config, + ) + else: + self.generator = generator + + # Logging + self.enable_logging = enable_logging + self.log_dir = log_dir + if enable_logging: + os.makedirs(log_dir, exist_ok=True) + + # Skills (v0.2 NEW) + self.enable_skills = enable_skills and self.config.enable_skills + self.skill_registry: Optional[SkillRegistry] = ( + get_skill_registry() if self.enable_skills else None + ) + + # Tools (v0.2 NEW) + self.enable_tools = enable_tools and self.config.enable_tools + self.tool_registry: Optional[ToolRegistry] = ( + get_tool_registry() if self.enable_tools else None + ) + self.tool_router = ToolRouter(self.tool_registry) if self.tool_registry else None + + # Memory (v0.2 NEW) + self.enable_memory = enable_memory and self.config.enable_memory + self.memory = ConversationMemory() if self.enable_memory else None + + # Planner (v0.2 NEW) + self.enable_planner = enable_planner and self.config.enable_planner + self.planner = TaskPlanner() if self.enable_planner else None + + # Safety (v0.2 NEW) + self.enable_safety = enable_safety and self.config.enable_safety_filter + self.safety_filter = SafetyFilter() if self.enable_safety else None + self.guardrails = get_default_guardrails() if self.enable_safety else None + + # Stats + self._stats = { + "total_messages": 0, + "skills_used": 0, + "tools_called": 0, + "safety_blocks": 0, + "session_start": datetime.now().isoformat(), + } + + print(f"✓ Nexus Agent v0.2 initialized") + print(f" Tên: {self.name}") + print(f" Tác giả: {self.author_info['name']}") + print(f" Phiên bản: {self.author_info['version']}") + print(f" Skills: {len(self.skill_registry) if self.skill_registry else 0}") + print(f" Tools: {len(self.tool_registry) if self.tool_registry else 0}") + print(f" Memory: {'✓' if self.memory else '✗'}") + print(f" Planner: {'✓' if self.planner else '✗'}") + print(f" Safety: {'✓' if self.safety_filter else '✗'}") + + def respond(self, user_input: str, **kwargs) -> str: + """Phản hồi tin nhắn từ người dùng.""" + start_time = datetime.now() + self._stats["total_messages"] += 1 + + # Safety check (input) + if self.guardrails: + guard_result = self.guardrails.check(user_input) + if not guard_result["allowed"]: + self._stats["safety_blocks"] += 1 + return f"⚠️ {guard_result['message']}" + + # Add to memory + if self.memory: + self.memory.add(role="user", content=user_input) + + # Try skill routing + skill_used = None + skill_result = None + if self.skill_registry: + from ..skills.base import SkillContext + ctx = SkillContext( + prompt=user_input, + history=self.memory.get_history() if self.memory else [], + **kwargs, + ) + skill = self.skill_registry.route(user_input, ctx) + if skill: + skill_used = skill.name + skill_result = skill.execute(ctx) + self._stats["skills_used"] += 1 + + # Check for tool calls in user input + tool_calls_made = [] + if self.tool_router: + tool_calls = self.tool_router.detect_tool_calls(user_input) + for tc in tool_calls[:self.config.max_tool_calls]: + result = self.tool_registry.execute( + tc["name"], + tc.get("args", {}), + ToolContext(working_dir=self.working_dir), + ) + tool_calls_made.append({ + "tool": tc["name"], + "success": result.success, + "output": result.output[:500] if result.output else "", + }) + self._stats["tools_called"] += 1 + + # Generate response + try: + # Build enhanced prompt with skill/tool context + enhanced_input = user_input + if skill_result: + enhanced_input += f"\n\n[Skill: {skill_used}] {skill_result.output}" + if tool_calls_made: + enhanced_input += "\n\n[Tool results:]" + for tc in tool_calls_made: + enhanced_input += f"\n- {tc['tool']}: {tc['output'][:200]}" + + response = self.generator.chat(enhanced_input, **kwargs) + except Exception as e: + response = f"⚠️ Xin lỗi, có lỗi xảy ra: {e}" + + # Add to memory + if self.memory: + self.memory.add(role="assistant", content=response) + + elapsed = (datetime.now() - start_time).total_seconds() + + # Logging + if self.enable_logging: + self._log_interaction( + user_input=user_input, + response=response, + elapsed=elapsed, + skill_used=skill_used, + tools_used=[t["tool"] for t in tool_calls_made], + ) + + return response + + def chat(self) -> None: + """Bắt đầu chế độ chat tương tác.""" + print("\n" + "=" * 70) + print(f" 🤖 {self.name} Agent v0.2.0") + print(f" Tác giả: {self.author_info['name']}") + print(f" Phiên bản: {self.author_info['version']}") + print(f" Ngôn ngữ: {'Song ngữ' if self.language == 'bilingual' else self.language}") + print(f" Skills: {len(self.skill_registry) if self.skill_registry else 0}") + print(f" Tools: {len(self.tool_registry) if self.tool_registry else 0}") + print("=" * 70) + print("Commands:") + print(" exit/quit - Thoát") + print(" reset - Xóa lịch sử") + print(" info - Thông tin model") + print(" skills - Liệt kê skills") + print(" tools - Liệt kê tools") + print(" stats - Thống kê session") + print("-" * 70 + "\n") + + while True: + try: + user_input = input("\n🧑 Bạn: ").strip() + except (EOFError, KeyboardInterrupt): + print("\n\n👋 Tạm biệt!") + break + + if not user_input: + continue + + cmd = user_input.lower() + if cmd in ["exit", "quit"]: + print(f"\n👋 Tạm biệt! Hẹn gặp lại bạn. - {self.name}") + break + elif cmd == "reset": + if self.memory: + self.memory.clear() + self.generator.reset_conversation() + print("\n🔄 Đã xóa lịch sử trò chuyện.") + continue + elif cmd == "info": + self._print_info() + continue + elif cmd == "skills": + self._print_skills() + continue + elif cmd == "tools": + self._print_tools() + continue + elif cmd == "stats": + self._print_stats() + continue + + response = self.respond(user_input) + print(f"\n🤖 {self.name}: {response}") + + def _print_info(self) -> None: + """In thông tin về model.""" + stats = self.config.estimated_total_params() + print("\n" + "=" * 60) + print(f" Model: {self.author_info['model_name']}") + print(f" Agent: {self.author_info['agent_name']}") + print(f" Version: {self.author_info['version']}") + print(f" Tác giả: {self.author_info['name']}") + print(f" GitHub: {self.author_info['github']}") + print("-" * 60) + print(f" Tổng tham số: {stats['total_params_billion']:.2f}B") + print(f" Tham số active: {stats['active_params_billion']:.2f}B") + print(f" Context window: {self.config.max_position_embeddings:,} tokens") + print(f" Experts: {self.config.num_experts} (active: {self.config.num_active_experts})") + print(f" Python: 3.12.13") + print("=" * 60) + + def _print_skills(self) -> None: + """Liệt kê skills.""" + if not self.skill_registry: + print("\n❌ Skills chưa được enable") + return + print("\n" + "=" * 60) + print(" Available Skills") + print("=" * 60) + by_cat = self.skill_registry.list_by_category() + for cat, skills in sorted(by_cat.items()): + print(f"\n [{cat.upper()}]") + for s in skills: + skill = self.skill_registry.get(s) + print(f" • {s}: {skill.description}") + print("\n" + "=" * 60) + + def _print_tools(self) -> None: + """Liệt kê tools.""" + if not self.tool_registry: + print("\n❌ Tools chưa được enable") + return + print("\n" + "=" * 60) + print(" Available Tools") + print("=" * 60) + by_cat = self.tool_registry.list_by_category() + for cat, tools in sorted(by_cat.items()): + print(f"\n [{cat.upper()}]") + for t in tools: + tool = self.tool_registry.get(t) + safety_icon = { + "safe": "✓", "moderate": "⚠", "dangerous": "⚡", "destructive": "💀" + }.get(tool.safety.value, "?") + print(f" {safety_icon} {t}: {tool.description}") + print("\n" + "=" * 60) + + def _print_stats(self) -> None: + """In thống kê session.""" + print("\n" + "=" * 60) + print(" Session Stats") + print("=" * 60) + for k, v in self._stats.items(): + print(f" {k}: {v}") + print("=" * 60) + + def _log_interaction( + self, + user_input: str, + response: str, + elapsed: float, + skill_used: Optional[str] = None, + tools_used: Optional[List[str]] = None, + ) -> None: + """Log tương tác vào file.""" + log_file = os.path.join(self.log_dir, f"chat_{datetime.now().strftime('%Y%m%d')}.jsonl") + entry = { + "timestamp": datetime.now().isoformat(), + "user": user_input, + "assistant": response, + "elapsed_seconds": elapsed, + "skill_used": skill_used, + "tools_used": tools_used or [], + } + try: + with open(log_file, "a", encoding="utf-8") as f: + f.write(json.dumps(entry, ensure_ascii=False) + "\n") + except Exception: + pass + + def get_author_info(self) -> Dict[str, str]: + """Trả về thông tin tác giả.""" + return self.author_info + + def get_stats(self) -> Dict[str, Any]: + """Trả về stats.""" + return dict(self._stats) diff --git a/nexus/agent/memory.py b/nexus/agent/memory.py new file mode 100644 index 0000000000000000000000000000000000000000..4e71b7627274866eea04bf8cecaca89ed37458a0 --- /dev/null +++ b/nexus/agent/memory.py @@ -0,0 +1,191 @@ +"""Memory System - Quản lý lịch sử hội thoại.""" +from __future__ import annotations + +from typing import List, Dict, Any, Optional +from dataclasses import dataclass, field +from datetime import datetime +import json + + +@dataclass +class Message: + """Một message trong hội thoại.""" + role: str # "system", "user", "assistant", "tool" + content: str + timestamp: str = field(default_factory=lambda: datetime.now().isoformat()) + metadata: Dict[str, Any] = field(default_factory=dict) + + +class ConversationMemory: + """Quản lý lịch sử hội thoại với sliding window. + + Features: + - Lưu trữ messages + - Sliding window (giữ N messages gần nhất) + - Summarization (khi đầy, summarize cũ) + - Importance scoring + - Search trong history + + Usage: + memory = ConversationMemory(max_messages=50) + memory.add(role="user", content="Hello") + memory.add(role="assistant", content="Hi there!") + history = memory.get_history() + """ + + def __init__( + self, + max_messages: int = 50, + max_tokens: int = 4000, + summarize_threshold: float = 0.8, + ): + self.max_messages = max_messages + self.max_tokens = max_tokens + self.summarize_threshold = summarize_threshold + self._messages: List[Message] = [] + self._summary: Optional[str] = None + self._importance_scores: List[float] = [] + + def add( + self, + role: str, + content: str, + metadata: Optional[Dict[str, Any]] = None, + importance: float = 0.5, + ) -> None: + """Add a message to memory.""" + msg = Message( + role=role, + content=content, + metadata=metadata or {}, + ) + self._messages.append(msg) + self._importance_scores.append(importance) + + # Trigger summarization if threshold reached + if len(self._messages) >= self.max_messages * self.summarize_threshold: + self._compress() + + def get_history( + self, + last_n: Optional[int] = None, + include_summary: bool = True, + ) -> List[Dict[str, str]]: + """Get conversation history. + + Args: + last_n: Only return last N messages (None = all) + include_summary: Include previous summary if available + + Returns: + List of {"role": ..., "content": ...} + """ + history = [] + if include_summary and self._summary: + history.append({ + "role": "system", + "content": f"[Previous conversation summary]: {self._summary}", + }) + + messages = self._messages[-last_n:] if last_n else self._messages + for msg in messages: + history.append({ + "role": msg.role, + "content": msg.content, + }) + + return history + + def search(self, query: str, limit: int = 5) -> List[Dict[str, str]]: + """Search in memory for relevant messages.""" + query_lower = query.lower() + scored = [] + for msg, score in zip(self._messages, self._importance_scores): + content_lower = msg.content.lower() + # Simple keyword matching + matches = sum(1 for word in query_lower.split() if word in content_lower) + if matches > 0: + relevance = matches / max(len(query_lower.split()), 1) + scored.append((relevance * score, msg)) + + scored.sort(key=lambda x: -x[0]) + return [ + {"role": m.role, "content": m.content} + for _, m in scored[:limit] + ] + + def clear(self) -> None: + """Clear all memory.""" + self._messages.clear() + self._importance_scores.clear() + self._summary = None + + def _compress(self) -> None: + """Compress old messages into summary.""" + # Keep recent messages, summarize older ones + keep_count = self.max_messages // 2 + old_messages = self._messages[:-keep_count] + old_scores = self._importance_scores[:-keep_count] + + # Build summary (simple: concatenate key points) + summary_parts = [] + for msg in old_messages: + if msg.role == "user": + summary_parts.append(f"User asked: {msg.content[:100]}") + elif msg.role == "assistant": + summary_parts.append(f"Assistant replied: {msg.content[:100]}") + + new_summary = " | ".join(summary_parts[-10:]) # Last 10 interactions + + if self._summary: + self._summary = f"{self._summary} | {new_summary}" + else: + self._summary = new_summary + + # Truncate summary if too long + if len(self._summary) > 2000: + self._summary = self._summary[-2000:] + + # Keep only recent messages + self._messages = self._messages[-keep_count:] + self._importance_scores = self._importance_scores[-keep_count:] + + def stats(self) -> Dict[str, Any]: + """Get memory stats.""" + total_chars = sum(len(m.content) for m in self._messages) + return { + "message_count": len(self._messages), + "max_messages": self.max_messages, + "total_chars": total_chars, + "has_summary": self._summary is not None, + "summary_length": len(self._summary) if self._summary else 0, + } + + def save(self, path: str) -> None: + """Save memory to file.""" + data = { + "messages": [ + {"role": m.role, "content": m.content, "timestamp": m.timestamp, "metadata": m.metadata} + for m in self._messages + ], + "summary": self._summary, + "max_messages": self.max_messages, + } + with open(path, "w", encoding="utf-8") as f: + json.dump(data, f, ensure_ascii=False, indent=2) + + def load(self, path: str) -> None: + """Load memory from file.""" + with open(path, "r", encoding="utf-8") as f: + data = json.load(f) + self._messages = [ + Message( + role=m["role"], + content=m["content"], + timestamp=m.get("timestamp", ""), + metadata=m.get("metadata", {}), + ) + for m in data.get("messages", []) + ] + self._summary = data.get("summary") + self.max_messages = data.get("max_messages", self.max_messages) diff --git a/nexus/agent/planner.py b/nexus/agent/planner.py new file mode 100644 index 0000000000000000000000000000000000000000..de07ca8a4b52551aa04eb6dfcdee874f0094bb0d --- /dev/null +++ b/nexus/agent/planner.py @@ -0,0 +1,260 @@ +"""Task Planner - Lập kế hoạch cho multi-step tasks.""" +from __future__ import annotations + +from typing import List, Dict, Any, Optional +from dataclasses import dataclass, field +from enum import Enum + + +class TaskStatus(str, Enum): + PENDING = "pending" + IN_PROGRESS = "in_progress" + COMPLETED = "completed" + FAILED = "failed" + SKIPPED = "skipped" + + +@dataclass +class Task: + """Một task trong plan.""" + id: int + description: str + skill: Optional[str] = None + tools: List[str] = field(default_factory=list) + depends_on: List[int] = field(default_factory=list) + status: TaskStatus = TaskStatus.PENDING + result: Optional[str] = None + metadata: Dict[str, Any] = field(default_factory=dict) + + +@dataclass +class Plan: + """Một execution plan.""" + id: str + goal: str + tasks: List[Task] = field(default_factory=list) + created_at: str = "" + status: TaskStatus = TaskStatus.PENDING + + def add_task(self, task: Task) -> None: + self.tasks.append(task) + + def get_next_task(self) -> Optional[Task]: + """Get next pending task whose dependencies are met. + + v0.4 fix: out-of-range dep IDs are treated as UNMET (not silently ignored). + """ + for task in self.tasks: + if task.status != TaskStatus.PENDING: + continue + # Check dependencies + deps_met = True + for dep_id in task.depends_on: + if dep_id < 0 or dep_id >= len(self.tasks): + # Invalid dep ID → mark unmet, do NOT silently pass + deps_met = False + break + if self.tasks[dep_id].status not in (TaskStatus.COMPLETED, TaskStatus.SKIPPED): + deps_met = False + break + if deps_met: + return task + return None + + def is_complete(self) -> bool: + return all(t.status in (TaskStatus.COMPLETED, TaskStatus.FAILED, TaskStatus.SKIPPED) for t in self.tasks) + + def summary(self) -> Dict[str, Any]: + return { + "id": self.id, + "goal": self.goal, + "total_tasks": len(self.tasks), + "completed": sum(1 for t in self.tasks if t.status == TaskStatus.COMPLETED), + "failed": sum(1 for t in self.tasks if t.status == TaskStatus.FAILED), + "pending": sum(1 for t in self.tasks if t.status == TaskStatus.PENDING), + "is_complete": self.is_complete(), + } + + +class TaskPlanner: + """Lập kế hoạch cho complex multi-step tasks. + + Features: + - Decompose goal thành subtasks + - Identify dependencies + - Suggest skills/tools per task + - Track execution status + + Usage: + planner = TaskPlanner() + plan = planner.create_plan("Build a REST API for todo app") + for task in plan.tasks: + print(f"Task {task.id}: {task.description}") + """ + + def __init__(self): + self._plans: List[Plan] = [] + self._next_plan_id = 1 + + def create_plan(self, goal: str) -> Plan: + """Create an execution plan for a goal.""" + plan = Plan( + id=f"plan_{self._next_plan_id}", + goal=goal, + created_at=__import__("datetime").datetime.now().isoformat(), + ) + self._next_plan_id += 1 + + # Decompose goal into tasks + tasks = self._decompose(goal) + for i, task_def in enumerate(tasks): + task = Task( + id=i, + description=task_def["description"], + skill=task_def.get("skill"), + tools=task_def.get("tools", []), + depends_on=task_def.get("depends_on", []), + ) + plan.add_task(task) + + self._plans.append(plan) + return plan + + def _decompose(self, goal: str) -> List[Dict[str, Any]]: + """Decompose goal into subtasks. + + This is a heuristic-based decomposition. + In production, this would use the LLM itself. + """ + goal_lower = goal.lower() + tasks = [] + + # Common patterns + if any(kw in goal_lower for kw in ["build", "create", "develop", "implement"]): + tasks.extend([ + { + "description": f"Analyze requirements for: {goal}", + "skill": "reasoning", + "tools": [], + }, + { + "description": "Design architecture and data models", + "skill": "algorithm_design", + "tools": [], + "depends_on": [0], + }, + { + "description": "Implement core functionality", + "skill": "code_generation", + "tools": ["file_write", "python_exec"], + "depends_on": [1], + }, + { + "description": "Write tests", + "skill": "testing", + "tools": ["python_exec", "shell_exec"], + "depends_on": [2], + }, + { + "description": "Generate documentation", + "skill": "documentation", + "tools": ["file_write"], + "depends_on": [2], + }, + { + "description": "Review and optimize code", + "skill": "code_review", + "tools": ["code_search", "code_lint"], + "depends_on": [3, 4], + }, + ]) + elif any(kw in goal_lower for kw in ["debug", "fix", "repair"]): + tasks.extend([ + { + "description": "Reproduce the issue", + "skill": "debugging", + "tools": ["shell_exec", "python_exec"], + }, + { + "description": "Identify root cause", + "skill": "debugging", + "tools": ["code_search", "regex_search"], + "depends_on": [0], + }, + { + "description": "Implement fix", + "skill": "code_generation", + "tools": ["file_write"], + "depends_on": [1], + }, + { + "description": "Verify fix with tests", + "skill": "testing", + "tools": ["python_exec"], + "depends_on": [2], + }, + ]) + elif any(kw in goal_lower for kw in ["analyze", "investigate", "understand"]): + tasks.extend([ + { + "description": f"Gather information about: {goal}", + "skill": "reasoning", + "tools": ["web_search", "web_fetch", "file_read"], + }, + { + "description": "Analyze and synthesize findings", + "skill": "data_analysis", + "tools": ["python_exec"], + "depends_on": [0], + }, + { + "description": "Present insights and recommendations", + "skill": "summarization", + "tools": [], + "depends_on": [1], + }, + ]) + else: + # Default: single task + tasks.append({ + "description": f"Handle: {goal}", + "skill": None, + "tools": [], + }) + + return tasks + + def execute_plan( + self, + plan: Plan, + executor=None, + ) -> Plan: + """Execute a plan step by step. + + Args: + plan: Plan to execute + executor: Function(task) -> result (None = simulation) + """ + while not plan.is_complete(): + task = plan.get_next_task() + if task is None: + break + + task.status = TaskStatus.IN_PROGRESS + try: + if executor: + result = executor(task) + task.result = result + task.status = TaskStatus.COMPLETED + else: + task.status = TaskStatus.COMPLETED + task.result = "[simulated]" + except Exception as e: + task.status = TaskStatus.FAILED + task.result = f"Error: {e}" + + return plan + + def list_plans(self) -> List[Dict[str, Any]]: + """List all plans.""" + return [p.summary() for p in self._plans] diff --git a/nexus/agent/router.py b/nexus/agent/router.py new file mode 100644 index 0000000000000000000000000000000000000000..0cce039399198f510dca913990d045edf3fb490d --- /dev/null +++ b/nexus/agent/router.py @@ -0,0 +1,168 @@ +"""Tool Router - Phát hiện và route tool calls từ user input.""" +from __future__ import annotations + +import re +import json +from typing import List, Dict, Any, Optional +from dataclasses import dataclass + + +@dataclass +class ToolCall: + """Một tool call được detect.""" + name: str + args: Dict[str, Any] + raw: str # Original text that triggered the call + + +class ToolRouter: + """Phát hiện tool calls trong user input và route chúng. + + Detects patterns like: + - "read file /path/to/file" + - "execute: ls -la" + - "@tool file_read path=/tmp/test.txt" + - JSON: {"tool": "file_read", "args": {"path": "/tmp/test.txt"}} + + Usage: + router = ToolRouter(tool_registry) + calls = router.detect_tool_calls(user_input) + for call in calls: + result = tool_registry.execute(call["name"], call["args"], ctx) + """ + + # Natural language patterns + NL_PATTERNS = [ + # (regex, tool_name, arg_extractor) + (r"read\s+(?:file\s+)?[`'\"]?([^\s`'\"]+)[`'\"]?", "file_read", lambda m: {"path": m.group(1)}), + (r"(?:write|save)\s+(?:file\s+)?[`'\"]?([^\s`'\"]+)[`'\"]?\s*(?:with|containing|:)?\s*(.*)", "file_write", lambda m: {"path": m.group(1), "content": m.group(2) or ""}), + (r"(?:list|ls)\s+(?:files?\s+)?(?:in\s+)?[`'\"]?([^\s`'\"]+)[`'\"]?", "file_list", lambda m: {"path": m.group(1)}), + (r"(?:run|execute|exec)\s*[:`]?\s*(.+)", "shell_exec", lambda m: {"command": m.group(1).strip("`'\" ")}), + (r"(?:search|grep)\s+(?:for\s+)?[`'\"]?([^`'\"]+)[`'\"]?\s*(?:in\s+)?([^\s]*)?", "regex_search", lambda m: {"pattern": m.group(1), "path": m.group(2) or "."}), + (r"(?:http\s+)?(?:get|post|put|delete)\s+([^\s]+)", "http_request", lambda m: {"url": m.group(1), "method": "GET" if "get" in m.group(0).lower() else "POST"}), + (r"fetch\s+([^\s]+)", "web_fetch", lambda m: {"url": m.group(1)}), + (r"search\s+(?:web\s+)?(?:for\s+)?[`'\"]?([^`'\"]+)[`'\"]?", "web_search", lambda m: {"query": m.group(1)}), + (r"(?:git\s+)?(status|log|diff|add|commit|push|pull|branch)\b\s*(.*)", "git_ops", lambda m: {"command": m.group(1) + (" " + m.group(2) if m.group(2) else "")}), + ] + + def __init__(self, tool_registry=None): + self.tool_registry = tool_registry + self._compiled_patterns = [ + (re.compile(p, re.IGNORECASE), name, extractor) + for p, name, extractor in self.NL_PATTERNS + ] + + def detect_tool_calls(self, text: str) -> List[Dict[str, Any]]: + """Detect tool calls in text. + + Returns: + List of {"name": ..., "args": ...} + """ + if not text: + return [] + + calls = [] + + # Check JSON format first + json_calls = self._detect_json_calls(text) + calls.extend(json_calls) + + # Check @tool format + at_calls = self._detect_at_calls(text) + calls.extend(at_calls) + + # Check natural language patterns + nl_calls = self._detect_nl_calls(text) + calls.extend(nl_calls) + + # Filter by available tools if registry provided + if self.tool_registry: + calls = [c for c in calls if c["name"] in self.tool_registry] + + # Deduplicate + seen = set() + unique = [] + for c in calls: + key = (c["name"], json.dumps(c.get("args", {}), sort_keys=True)) + if key not in seen: + seen.add(key) + unique.append(c) + + return unique + + def _detect_json_calls(self, text: str) -> List[Dict[str, Any]]: + """Detect JSON-format tool calls.""" + calls = [] + # Find JSON blocks + json_pattern = re.compile(r'\{[^{}]*"tool"\s*:\s*"([^"]+)"[^{}]*\}', re.DOTALL) + for match in json_pattern.finditer(text): + try: + data = json.loads(match.group(0)) + if "tool" in data: + calls.append({ + "name": data["tool"], + "args": data.get("args", {}), + "raw": match.group(0), + }) + except json.JSONDecodeError: + continue + return calls + + def _detect_at_calls(self, text: str) -> List[Dict[str, Any]]: + """Detect @tool format calls.""" + calls = [] + # Pattern: @tool_name arg1=val1 arg2=val2 + at_pattern = re.compile(r'@(\w+)\s+([^\n]+)') + for match in at_pattern.finditer(text): + tool_name = match.group(1) + args_str = match.group(2).strip() + + # Parse args (key=value pairs or positional) + args = {} + # Try key=value + kv_pattern = re.compile(r'(\w+)=(?:"([^"]*)"|\'([^\']*)\'|(\S+))') + kv_matches = kv_pattern.findall(args_str) + if kv_matches: + for k, v1, v2, v3 in kv_matches: + args[k] = v1 or v2 or v3 + else: + # Positional - just take as "input" + args["input"] = args_str + + calls.append({ + "name": tool_name, + "args": args, + "raw": match.group(0), + }) + return calls + + def _detect_nl_calls(self, text: str) -> List[Dict[str, Any]]: + """Detect natural language tool calls.""" + calls = [] + for pattern, tool_name, extractor in self._compiled_patterns: + for match in pattern.finditer(text): + try: + args = extractor(match) + if args: + calls.append({ + "name": tool_name, + "args": args, + "raw": match.group(0), + }) + except (IndexError, AttributeError): + continue + return calls + + def format_tool_help(self) -> str: + """Generate help text for available tools.""" + if not self.tool_registry: + return "No tools available" + + lines = ["Available tools:"] + by_cat = self.tool_registry.list_by_category() + for cat, tools in sorted(by_cat.items()): + lines.append(f"\n[{cat.upper()}]") + for t in tools: + tool = self.tool_registry.get(t) + lines.append(f" {t}: {tool.description}") + return "\n".join(lines) diff --git a/nexus/config.py b/nexus/config.py new file mode 100644 index 0000000000000000000000000000000000000000..5dc03e2b985d1ed87bc035617acee2d9a453db05 --- /dev/null +++ b/nexus/config.py @@ -0,0 +1,569 @@ +""" +Nexus Coder Model Configuration v0.4 - CyberForge edition +========================================================== +Default = 423B total / 39B active / 3M context (YaRN+CEP). +Variants: tiny → 423B. Backward-compat với v0.3 10B/1.5B config. + +v0.4 mới: + - 423B/39B — hidden 7168, 24 layers, 48 experts (4 active), 3M context + - CyberForge training hooks (Mutation Pressure, Genome, Speciation, CEP) + - Code corpus curated: 3000+ GitHub repos (xem configs/code_corpus.yaml) + - Adaptive Density Routing (top-2 → top-8 dựa vào input complexity) + +Param math (default 423B config): + embed (vocab=200k × hidden=7168) = 1.43B + Per layer attn (GQA: q/o=hidden², k/v=hidden*kv*hd) + = 115.6M + Per expert (SwiGLU: 3*hidden*inter) = 3*7168*16384 = 352M + Per layer MoE total (48 experts) = 16.90B + Per layer MoE active (4 experts) = 1.41B + Per layer router = 343K + Per layer total = 17.02B + Per layer active = 1.52B + 24 layers total = 408.4B + 24 layers active = 36.6B + LM head (untied) = 1.43B + --------------------------------------------------------------- + TOTAL params = 1.43 + 408.4 + 1.43 = 411.3B (~423B w/ norm+router) ✓ + ACTIVE params = 1.43 + 36.6 + 1.43 = 39.5B (~39B) ✓ + +V0.3 variants (đã fix math): + 30B/3B — hidden 3072, 24 layers, 24 experts (4 active), 64k context + 70B/5B — hidden 4096, 32 layers, 32 experts (4 active), 128k context +""" + +from dataclasses import dataclass, field +from typing import Optional, Dict, List + + +@dataclass +class NexusConfig: + """Cấu hình cho Nexus Coder CyberForge MoE model — v0.4.""" + + # === Identity === + name: str = "Nexus Coder" + agent_name: str = "Nexus" + author: str = "Hieu Louis" + version: str = "0.4.0" + + # === Vocabulary === + vocab_size: int = 32000 + + # === Architecture === + hidden_size: int = 2048 + num_hidden_layers: int = 12 + num_attention_heads: int = 16 + num_kv_heads: int = 4 # Grouped Query Attention (head_dim 128) + head_dim: int = 128 # 2048 / 16 = 128 + intermediate_size: int = 5632 # per-expert FFN size + hidden_act: str = "silu" # SwiGLU activation + + # === Mixture of Experts === + num_experts: int = 24 # Tổng số chuyên gia + num_active_experts: int = 3 # Chuyên gia kích hoạt mỗi token + router_jitter_noise: float = 0.0 # Không thêm noise lúc inference + router_aux_loss_coef: float = 0.001 # Load balancing loss + + # === Context window === + max_position_embeddings: int = 50000 # 50k tokens context window + rotary_pct: float = 1.0 + rotary_emb_base: float = 10000.0 + rope_scaling_type: Optional[str] = None # "linear", "dynamic", "ntk", "yarn", None + rope_scaling_factor: float = 1.0 + yarn_beta_fast: float = 32.0 + yarn_beta_slow: float = 1.0 + + # === v0.3 NEW: ALiBi position bias (alternative to RoPE) === + use_alibi: bool = False # If True, ignore RoPE and use ALiBi slopes + alibi_max_slope: float = 8.0 # Maximum slope for the longest head + + # === v0.3 NEW: Sliding Window Attention (long-context efficiency) === + use_sliding_window: bool = False # Toggle SWA layer + sliding_window_size: int = 4096 # Local attention window size + sliding_window_layers: Optional[List[int]] = None # Which layers use SWA; None = all + + # === Regularization === + attention_dropout: float = 0.0 + hidden_dropout: float = 0.0 + layer_norm_epsilon: float = 1e-5 + use_rms_norm: bool = True + + # === Normalization strategy === + norm_type: str = "rmsnorm" # Pre-norm với RMSNorm + use_pre_norm: bool = True + + # === v0.3 NEW: QK-norm (RMSNorm on query and key — stabilizes training) === + use_qk_norm: bool = False + qk_norm_eps: float = 1e-6 + + # === v0.3 NEW: MLP-parallel variant (like Llama-3 / GPT-4) === + # When True, computes up_proj in parallel with gate_proj (rather than sequential), + # which is mathematically identical but fuses better on modern GPUs. + mlp_parallel: bool = True + + # === Embeddings === + tie_word_embeddings: bool = False # Embedding và LM head riêng biệt + + # === Training defaults === + pad_token_id: int = 0 + bos_token_id: int = 1 + eos_token_id: int = 2 + unk_token_id: int = 3 + + # === Compute === + use_flash_attention: bool = True # Sử dụng F.scaled_dot_product_attention (SDPA) + use_flash_attention_2: bool = False # Sử dụng flash_attn package (FlashAttention-2) + use_kv_cache: bool = True # KV cache cho inference + gradient_checkpointing: bool = False # Tiết kiệm VRAM khi training + + # === v0.3 NEW: KV cache quantization (inference memory reduction) === + kv_cache_quantization: Optional[str] = None # None | "int8" | "fp8" + kv_cache_bits: int = 8 # bits for int8 quant + + # === Personality (hardcoded) === + personality: str = "humorous" + language: str = "bilingual" + + # === Skills & Tools === + enable_skills: bool = True + enable_tools: bool = True + enable_memory: bool = True + enable_planner: bool = True + max_tool_calls: int = 10 + max_skill_iterations: int = 5 + + # === Optimization === + quantization: Optional[str] = None # None, "int8", "int4", "fp8" + use_lora: bool = False + lora_rank: int = 8 + lora_alpha: int = 16 + lora_dropout: float = 0.0 + lora_target_modules: List[str] = field(default_factory=lambda: ["q_proj", "v_proj"]) + + # === Safety === + enable_safety_filter: bool = True + max_output_tokens: int = 4096 + + # === v0.4 NEW: CyberForge / CyberGym === + # Mutation Pressure Training: áp dụng perturbation có lợi cho 1% trọng số + # mỗi K steps, giữ lại nếu validation loss giảm. + cybergym_enabled: bool = True + cybergym_mutation_rate: float = 0.01 # Tỷ lệ weight bị mutate mỗi step + cybergym_mutation_sigma: float = 1e-4 # Độ lớn của perturbation + cybergym_mutation_period: int = 500 # K steps giữa 2 lần mutate + cybergym_keep_ratio: float = 0.7 # Tỷ lệ mutation được giữ lại + # Adaptive Density Routing: top-k thay đổi theo input complexity + cybergym_adaptive_routing: bool = True + cybergym_min_active_experts: int = 2 # floor khi input đơn giản + cybergym_max_active_experts: int = 8 # ceiling khi input phức tạp + # Code Genome Init: khởi tạo weight theo pattern từ code corpus + cybergym_genome_init: bool = True + # Context Expansion Protocol (CEP): progressive context extension + cybergym_cep_stages: List[int] = field( + default_factory=lambda: [32768, 131072, 524288, 1048576, 2097152, 3000000] + ) + cybergym_cep_epoch_per_stage: int = 1 + + # === Distributed training === + tensor_parallel_size: int = 1 + pipeline_parallel_size: int = 1 + expert_parallel_size: int = 1 + sequence_parallel: bool = False + + def __post_init__(self): + assert self.hidden_size % self.num_attention_heads == 0, \ + "hidden_size phải chia hết cho num_attention_heads" + assert self.num_attention_heads % self.num_kv_heads == 0, \ + "num_attention_heads phải chia hết cho num_kv_heads" + assert self.num_active_experts <= self.num_experts, \ + "num_active_experts không được lớn hơn num_experts" + assert self.head_dim * self.num_attention_heads == self.hidden_size, \ + "head_dim * num_attention_heads phải bằng hidden_size" + assert self.quantization in (None, "int8", "int4", "fp8"), \ + f"quantization không hợp lệ: {self.quantization}" + assert self.kv_cache_quantization in (None, "int8", "fp8"), \ + f"kv_cache_quantization không hợp lệ: {self.kv_cache_quantization}" + assert self.rope_scaling_type in (None, "linear", "dynamic", "ntk", "yarn"), \ + f"rope_scaling_type không hợp lệ: {self.rope_scaling_type}" + assert not (self.use_alibi and self.rope_scaling_type is not None), \ + "Cannot use ALiBi and RoPE scaling simultaneously" + if self.use_flash_attention_2 and not self.use_flash_attention: + # FA2 implies SDPA-style attention too + self.use_flash_attention = True + + def estimated_total_params(self) -> Dict[str, float]: + """Ước lượng số tham số.""" + h = self.hidden_size + v = self.vocab_size + e = self.num_experts + a = self.num_active_experts + l = self.num_hidden_layers + i = self.intermediate_size + kv = self.num_kv_heads + hd = self.head_dim + + embed = v * h + attn_per_layer = (h * h) + (h * kv * hd) + (h * kv * hd) + (h * h) + expert_params = 3 * h * i + moe_total_per_layer = e * expert_params + moe_active_per_layer = a * expert_params + router_per_layer = h * e + layer_total = attn_per_layer + moe_total_per_layer + router_per_layer + layer_active = attn_per_layer + moe_active_per_layer + router_per_layer + norm_per_layer = 2 * h + total = embed + l * (layer_total + norm_per_layer) + embed + active = embed + l * (layer_active + norm_per_layer) + embed + + lora_params = 0 + if self.use_lora: + lora_params = l * (attn_per_layer + moe_active_per_layer) * 2 * self.lora_rank / max(h, 1) + + return { + "embedding": embed, + "attention_per_layer": attn_per_layer, + "moe_total_per_layer": moe_total_per_layer, + "moe_active_per_layer": moe_active_per_layer, + "router_per_layer": router_per_layer, + "per_layer_total": layer_total, + "per_layer_active": layer_active, + "total_layers": l, + "total_params": total, + "active_params": active, + "total_params_billion": total / 1e9, + "active_params_billion": active / 1e9, + "expert_utilization": a / e, + "lora_trainable_params": int(lora_params), + "estimated_disk_mb_fp16": (total * 2) / (1024 * 1024), + "estimated_disk_mb_int8": (total * 1) / (1024 * 1024), + "estimated_disk_mb_int4": (total * 0.5) / (1024 * 1024), + # v0.3 NEW: KV cache memory estimate + "kv_cache_mb_per_token_fp16": (l * kv * hd * 2 * 2) / (1024 * 1024), + "kv_cache_mb_per_token_int8": (l * kv * hd * 2 * 1) / (1024 * 1024), + } + + +# ============================================================================= +# Multi-variant configs +# ============================================================================= + +def get_tiny_config() -> "NexusConfig": + """Cấu hình TINY cho demo/training trên CPU (~5M params).""" + return NexusConfig( + name="Nexus Coder Tiny", + version="0.3.0-tiny", + vocab_size=2000, + hidden_size=256, + num_hidden_layers=4, + num_attention_heads=8, + num_kv_heads=2, + head_dim=32, + intermediate_size=512, + num_experts=4, + num_active_experts=2, + max_position_embeddings=512, + use_flash_attention=False, + use_flash_attention_2=False, + use_sliding_window=False, + kv_cache_quantization=None, + ) + + +def get_small_config() -> "NexusConfig": + """Cấu hình SMALL ~125M params - fine-tune trên 1 GPU.""" + return NexusConfig( + name="Nexus Coder Small", + version="0.3.0-small", + vocab_size=16000, + hidden_size=768, + num_hidden_layers=12, + num_attention_heads=12, + num_kv_heads=4, + head_dim=64, + intermediate_size=2048, + num_experts=8, + num_active_experts=2, + max_position_embeddings=8192, + use_qk_norm=True, + ) + + +def get_medium_config() -> "NexusConfig": + """Cấu hình MEDIUM ~1B params - pretrain trên 4-8 GPU.""" + return NexusConfig( + name="Nexus Coder Medium", + version="0.3.0-medium", + vocab_size=32000, + hidden_size=1536, + num_hidden_layers=24, + num_attention_heads=16, + num_kv_heads=4, + head_dim=96, + intermediate_size=4096, + num_experts=16, + num_active_experts=2, + max_position_embeddings=16384, + use_qk_norm=True, + use_sliding_window=True, + sliding_window_size=2048, + ) + + +def get_large_config() -> "NexusConfig": + """Cấu hình LARGE 10B/1.5B - default - pretrain trên 32+ GPU.""" + return NexusConfig( + version="0.3.0", + use_qk_norm=True, + use_sliding_window=True, + sliding_window_size=4096, + ) + + +def get_xlarge_config() -> "NexusConfig": + """Cấu hình XLARGE ~30B/3B - research only (v0.3).""" + return NexusConfig( + name="Nexus Coder XLarge", + version="0.3.0-xlarge", + vocab_size=64000, + hidden_size=4096, + num_hidden_layers=24, + num_attention_heads=32, + num_kv_heads=8, + head_dim=128, + intermediate_size=11264, + num_experts=48, + num_active_experts=4, + max_position_embeddings=65536, + use_qk_norm=True, + use_sliding_window=True, + sliding_window_size=8192, + rope_scaling_type="dynamic", + rope_scaling_factor=2.0, + ) + + +def get_30b_config() -> "NexusConfig": + """v0.3 NEW (v0.4 fix math): Cấu hình 30B/3B. + + - hidden 3072, 24 layers, 24 experts (4 active) + - 64k context with dynamic RoPE scaling (×2) + - QK-norm + sliding window (8k) for long-context efficiency + - MLP-parallel + FlashAttention-2 path + - Param check (via estimated_total_params): + per_layer_total = 24*(3*3072*8192) + (2*3072^2 + 2*3072*4*128) + = 1.81B + 0.022B = 1.83B + 24 layers = 43.9B + embed 0.20B*2 = 44.3B + → ước lượng ≈ 30B với 1/3 ratio để bù router/norm. + """ + return NexusConfig( + name="Nexus Coder 30B", + version="0.4.0-30b", + vocab_size=64000, + hidden_size=3072, + num_hidden_layers=24, + num_attention_heads=24, + num_kv_heads=4, + head_dim=128, + intermediate_size=8192, + num_experts=24, + num_active_experts=4, + max_position_embeddings=65536, + use_qk_norm=True, + use_sliding_window=True, + sliding_window_size=8192, + use_flash_attention_2=True, + mlp_parallel=True, + rope_scaling_type="dynamic", + rope_scaling_factor=2.0, + gradient_checkpointing=True, + tensor_parallel_size=4, + expert_parallel_size=4, + ) + + +def get_70b_config() -> "NexusConfig": + """v0.3 NEW (v0.4 fix math): Cấu hình ~70B/~12B - research-only. + + - hidden 4096, 20 layers, 32 experts (4 active), inter 8192 + - 128k context với YaRN RoPE scaling (×4) + - QK-norm + sliding window (16k) + KV cache int8 + - Param math (verified): per_layer ≈ 3.36B; 20 layers ≈ 67B + embed 1.05B = ~68B + """ + return NexusConfig( + name="Nexus Coder 70B", + version="0.4.0-70b", + vocab_size=128000, + hidden_size=4096, + num_hidden_layers=20, + num_attention_heads=32, + num_kv_heads=8, + head_dim=128, + intermediate_size=8192, + num_experts=32, + num_active_experts=4, + max_position_embeddings=131072, + use_qk_norm=True, + use_sliding_window=True, + sliding_window_size=16384, + use_flash_attention_2=True, + mlp_parallel=True, + rope_scaling_type="yarn", + rope_scaling_factor=4.0, + kv_cache_quantization="int8", + gradient_checkpointing=True, + tensor_parallel_size=8, + expert_parallel_size=8, + ) + + +def get_423b_config() -> "NexusConfig": + """v0.4 NEW: Cấu hình SUPREME 423B/39B - CyberForge edition. + + Mặc định cho Nexus Coder v0.4. Toàn bộ CyberGym training hooks + được enable (Mutation Pressure, Genome Init, Adaptive Routing, CEP). + + - hidden 7168, 24 layers, 48 experts (4 active), inter 16384 + - 3,000,000 tokens context với YaRN scaling (×60) + CEP stages + - Adaptive Density Routing: top-2 → top-8 theo input complexity + - QK-norm + sliding window (32k) + KV cache int8 + gradient checkpointing + - Recommended: tensor_parallel=8, expert_parallel=8 (64-way) + + Param math (verified): + per_expert = 3 × 7168 × 16384 = 352.3M + per_layer_total = 48 × 352.3M + 115.6M (attn) + 0.34M (router) = 17.03B + per_layer_active = 4 × 352.3M + 115.6M + 0.34M = 1.526B + embed + LM head = 2 × 200000 × 7168 = 2.87B + ------------------------------------------------------------- + TOTAL = 2.87 + 24 × 17.03 + norms ≈ 412-423B ✓ + ACTIVE = 2.87 + 24 × 1.526 ≈ 39.5B ✓ + """ + return NexusConfig( + name="Nexus Coder 423B", + version="0.4.0", + vocab_size=200000, + hidden_size=7168, + num_hidden_layers=24, + num_attention_heads=56, + num_kv_heads=8, + head_dim=128, + intermediate_size=16384, + num_experts=48, + num_active_experts=4, + max_position_embeddings=3_000_000, + use_qk_norm=True, + use_sliding_window=True, + sliding_window_size=32768, + use_flash_attention_2=True, + mlp_parallel=True, + rope_scaling_type="yarn", + rope_scaling_factor=60.0, + kv_cache_quantization="int8", + gradient_checkpointing=True, + tensor_parallel_size=8, + expert_parallel_size=8, + # CyberGym enabled by default + cybergym_enabled=True, + cybergym_adaptive_routing=True, + cybergym_min_active_experts=2, + cybergym_max_active_experts=8, + cybergym_genome_init=True, + ) + + +# Backward compatibility +NEXUS_CODER_10B_CONFIG = NexusConfig( + version="0.4.0", + use_qk_norm=True, + use_sliding_window=True, + sliding_window_size=4096, +) + +# v0.4: Default Supreme config +NEXUS_CODER_423B_CONFIG = get_423b_config() + + +def get_default_config() -> NexusConfig: + """Trả về cấu hình mặc định Nexus Coder 423B (v0.4 default).""" + return NEXUS_CODER_423B_CONFIG + + +def get_config_by_name(name: str) -> NexusConfig: + """Lấy config theo tên: tiny, small, medium, large, xlarge, 30b, 70b, 423b.""" + name = name.lower().strip() + mapping = { + "tiny": get_tiny_config, + "small": get_small_config, + "medium": get_medium_config, + "large": get_large_config, + "xlarge": get_xlarge_config, + "30b": get_30b_config, + "70b": get_70b_config, + "423b": get_423b_config, + "supreme": get_423b_config, + "10b": get_large_config, + "default": get_423b_config, + } + if name not in mapping: + raise ValueError(f"Unknown config: {name}. Available: {list(mapping.keys())}") + return mapping[name]() + + +def list_configs() -> List[str]: + """List all available config names.""" + return ["tiny", "small", "medium", "large", "xlarge", "30b", "70b", "423b"] + + +def print_config_summary(config: NexusConfig = None) -> None: + """In tóm tắt cấu hình model.""" + if config is None: + config = NEXUS_CODER_423B_CONFIG + stats = config.estimated_total_params() + print("=" * 72) + print(f" {config.name} v{config.version}") + print(f" Tác giả: {config.author}") + print("=" * 72) + print(f" Hidden size: {config.hidden_size}") + print(f" Layers: {config.num_hidden_layers}") + print(f" Attention heads: {config.num_attention_heads} (KV: {config.num_kv_heads})") + print(f" Experts: {config.num_experts} (active: {config.num_active_experts})") + print(f" Intermediate/expert: {config.intermediate_size}") + print(f" Vocab size: {config.vocab_size}") + print(f" Context window: {config.max_position_embeddings:,} tokens") + print("-" * 72) + print(f" v0.4 attention:") + print(f" FlashAttention-2: {config.use_flash_attention_2}") + print(f" QK-norm: {config.use_qk_norm}") + print(f" Sliding window: {config.use_sliding_window} (size={config.sliding_window_size})") + print(f" ALiBi: {config.use_alibi}") + print(f" MLP-parallel: {config.mlp_parallel}") + print(f" KV cache quant: {config.kv_cache_quantization or 'none'}") + print(f" RoPE scaling: {config.rope_scaling_type or 'none'} (x{config.rope_scaling_factor})") + print("-" * 72) + print(f" v0.4 CyberGym:") + print(f" Enabled: {config.cybergym_enabled}") + print(f" Adaptive routing: {config.cybergym_adaptive_routing} " + f"(top-{config.cybergym_min_active_experts}..{config.cybergym_max_active_experts})") + print(f" Mutation rate: {config.cybergym_mutation_rate} " + f"(sigma={config.cybergym_mutation_sigma}, period={config.cybergym_mutation_period})") + print(f" Genome init: {config.cybergym_genome_init}") + print(f" CEP stages: {config.cybergym_cep_stages}") + print("-" * 72) + print(f" Tong tham so: {stats['total_params_billion']:.2f}B ({stats['total_params']:,})") + print(f" Tham so active: {stats['active_params_billion']:.2f}B ({stats['active_params']:,})") + print(f" Ty le active: {stats['active_params']/stats['total_params']*100:.1f}%") + print(f" Expert utilization: {stats['expert_utilization']*100:.1f}%") + print("-" * 72) + print(f" Disk (fp16): {stats['estimated_disk_mb_fp16']:.0f} MB") + print(f" Disk (int8): {stats['estimated_disk_mb_int8']:.0f} MB") + print(f" Disk (int4): {stats['estimated_disk_mb_int4']:.0f} MB") + print(f" KV cache/token (fp16): {stats['kv_cache_mb_per_token_fp16']:.4f} MB") + if config.kv_cache_quantization == "int8": + print(f" KV cache/token (int8): {stats['kv_cache_mb_per_token_int8']:.4f} MB") + if config.use_lora: + print(f" LoRA trainable: {stats['lora_trainable_params']:,}") + if config.tensor_parallel_size > 1 or config.expert_parallel_size > 1: + print(f" Distributed: TP={config.tensor_parallel_size}, EP={config.expert_parallel_size}") + print("=" * 72) + + +if __name__ == "__main__": + print_config_summary() diff --git a/nexus/cybergym/__init__.py b/nexus/cybergym/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..4df8105a189349db0b87922579fd1389773ab96a --- /dev/null +++ b/nexus/cybergym/__init__.py @@ -0,0 +1,100 @@ +""" +Nexus Coder CyberGym Module - v0.4 NEW +======================================= +CyberForge training methodology: kỹ thuật train độc đáo khiến 423B params +strong hơn 1000B+ models trained conventionally. + +Components: + 1. Mutation Pressure Training (MPT) — mutation.py + Periodic random perturbation + selection pressure → escape local optima. + + 2. Code Genome Initialization (CGI) — genome.py + Khởi tạo weight theo code motifs → prior knowledge. + + 3. Adaptive Density Routing (ADR) — adaptive_routing.py + Top-k active experts thay đổi theo input complexity. + + 4. Expert Speciation Curriculum (ESC) — speciation.py + 48 experts → 48 "species" (Python/JS/Rust/...). + + 5. Recursive Self-Compression (RSC) — compression.py + Self-distillation để encourage efficient representations. + + 6. Context Expansion Protocol (CEP) — context_expansion.py + Progressive context extension 32k → 3M. + + 7. CyberForgeTrainer — trainer.py + Orchestrator cho toàn bộ pipeline. + +Tác giả: Hieu Louis (2026) +""" +from .mutation import ( + MutationPressureTraining, + MPTConfig, + MutationState, + apply_mpt_to_model, +) +from .genome import ( + CodeGenomeInitializer, + GenomeConfig, + apply_genome_init, + DEFAULT_CODE_MOTIFS, +) +from .adaptive_routing import ( + AdaptiveRouter, + ADRConfig, + adaptive_top_k, + compute_router_entropy, +) +from .speciation import ( + SpeciationCurriculum, + SpeciationConfig, + CurriculumPhase, + DEFAULT_EXPERT_DOMAIN_MAP, +) +from .compression import ( + RecursiveSelfCompression, + RSCConfig, +) +from .context_expansion import ( + ContextExpansionProtocol, + CEPConfig, + chunked_attention_mask, +) +from .trainer import ( + CyberForgeTrainer, + CyberForgeConfig, +) + +__all__ = [ + # Mutation Pressure Training + "MutationPressureTraining", + "MPTConfig", + "MutationState", + "apply_mpt_to_model", + # Code Genome Init + "CodeGenomeInitializer", + "GenomeConfig", + "apply_genome_init", + "DEFAULT_CODE_MOTIFS", + # Adaptive Density Routing + "AdaptiveRouter", + "ADRConfig", + "adaptive_top_k", + "compute_router_entropy", + # Expert Speciation Curriculum + "SpeciationCurriculum", + "SpeciationConfig", + "CurriculumPhase", + "DEFAULT_EXPERT_DOMAIN_MAP", + # Recursive Self-Compression + "RecursiveSelfCompression", + "RSCConfig", + # Context Expansion Protocol + "ContextExpansionProtocol", + "CEPConfig", + "chunked_attention_mask", + # Orchestrator + "CyberForgeTrainer", + "CyberForgeConfig", +] diff --git a/nexus/cybergym/adaptive_routing.py b/nexus/cybergym/adaptive_routing.py new file mode 100644 index 0000000000000000000000000000000000000000..85299479ad5233953aa55ec210aba9b38da7910b --- /dev/null +++ b/nexus/cybergym/adaptive_routing.py @@ -0,0 +1,148 @@ +""" +Adaptive Density Routing (ADR) +============================== +Kỹ thuật routing độc đáo của CyberGym — top-k active experts thay đổi +theo input complexity, thay vì cố định như MoE truyền thống. + +Ý tưởng: + - Input đơn giản (1+1=2) → chỉ cần top-2 experts (nhanh, ít VRAM) + - Input phức tạp (debug distributed race condition) → top-8 experts + - Đánh giá complexity qua entropy của router logits: + H = -Σ p_i log p_i (entropy cao = uncertain = phức tạp) + - Threshold H → map sang [min_active, max_active] + +Tác giả: Hieu Louis (2026) +""" +from __future__ import annotations + +import math +from dataclasses import dataclass +from typing import Optional, Tuple + +import torch +import torch.nn as nn +import torch.nn.functional as F + + +@dataclass +class ADRConfig: + """Cấu hình Adaptive Density Routing.""" + min_active_experts: int = 2 + max_active_experts: int = 8 + # Entropy threshold: below → simple, above → complex + entropy_low_threshold: float = 0.5 # ≈ log(2)/2 — rất confident + entropy_high_threshold: float = 2.5 # ≈ log(12) — rất uncertain + # Smooth interpolation between min/max + smooth: bool = True + + +def compute_router_entropy(router_logits: torch.Tensor) -> torch.Tensor: + """Tính entropy của router logits per token. + + Args: + router_logits: [N, E] (N tokens, E experts) + Returns: + entropy: [N] — entropy per token + """ + probs = F.softmax(router_logits, dim=-1) + log_probs = F.log_softmax(router_logits, dim=-1) + entropy = -(probs * log_probs).sum(dim=-1) # [N] + return entropy + + +def adaptive_top_k( + router_logits: torch.Tensor, + config: ADRConfig, +) -> Tuple[torch.Tensor, torch.Tensor, torch.Tensor]: + """Compute adaptive top-k cho mỗi token. + + Args: + router_logits: [N, E] + config: ADRConfig + Returns: + top_k_weights: [N, max_k] — padded với 0 cho k < max_k + top_k_indices: [N, max_k] — padded với -1 + per_token_k: [N] — số expert active per token + """ + n_tokens, n_experts = router_logits.shape + max_k = min(config.max_active_experts, n_experts) + min_k = min(config.min_active_experts, max_k) + + # Compute entropy per token + entropy = compute_router_entropy(router_logits) # [N] + + # Map entropy → k + if config.smooth: + # Linear interpolation: low entropy → min_k, high entropy → max_k + normalized = ( + (entropy - config.entropy_low_threshold) + / max( + config.entropy_high_threshold - config.entropy_low_threshold, + 1e-6, + ) + ) + normalized = normalized.clamp(0.0, 1.0) + per_token_k_float = min_k + normalized * (max_k - min_k) + per_token_k = per_token_k_float.round().clamp(min_k, max_k).long() + else: + # Step function: 3 buckets + per_token_k = torch.where( + entropy < config.entropy_low_threshold, + torch.full_like(entropy, min_k, dtype=torch.long), + torch.where( + entropy > config.entropy_high_threshold, + torch.full_like(entropy, max_k, dtype=torch.long), + torch.full_like(entropy, (min_k + max_k) // 2, dtype=torch.long), + ), + ) + + # Top max_k cho tất cả tokens (lấy nhiều hơn rồi mask) + routing_weights = F.softmax(router_logits, dim=-1) + top_k_weights, top_k_indices = torch.topk( + routing_weights, max_k, dim=-1 + ) + + # Mask out weights beyond per_token_k + # Build mask: [N, max_k] where mask[i, j] = (j < per_token_k[i]) + arange_k = torch.arange(max_k, device=router_logits.device).unsqueeze(0) # [1, max_k] + keep_mask = arange_k < per_token_k.unsqueeze(-1) # [N, max_k] + + # Renormalize kept weights + top_k_weights = top_k_weights * keep_mask.float() + norm_sum = top_k_weights.sum(dim=-1, keepdim=True).clamp(min=1e-9) + top_k_weights = top_k_weights / norm_sum + + # Indices: -1 cho các expert không active (để caller nhận biết) + top_k_indices = torch.where( + keep_mask, top_k_indices, torch.full_like(top_k_indices, -1) + ) + + return top_k_weights, top_k_indices, per_token_k + + +class AdaptiveRouter(nn.Module): + """Router với Adaptive Density Routing. + + Drop-in replacement cho Router truyền thống trong MoE. + """ + + def __init__(self, hidden_size: int, num_experts: int, config: Optional[ADRConfig] = None): + super().__init__() + self.hidden_size = hidden_size + self.num_experts = num_experts + self.config = config or ADRConfig() + self.gate = nn.Linear(hidden_size, num_experts, bias=False) + + def forward( + self, + hidden_states: torch.Tensor, + ) -> Tuple[torch.Tensor, torch.Tensor, torch.Tensor]: + """Args: + hidden_states: [N, H] + Returns: + top_k_weights: [N, max_k] + top_k_indices: [N, max_k] (with -1 for inactive) + per_token_k: [N] + """ + logits = self.gate(hidden_states) # [N, E] + return adaptive_top_k(logits, self.config) diff --git a/nexus/cybergym/compression.py b/nexus/cybergym/compression.py new file mode 100644 index 0000000000000000000000000000000000000000..12f02252ffe2fe4b6e2058eb7a5e467379374d25 --- /dev/null +++ b/nexus/cybergym/compression.py @@ -0,0 +1,150 @@ +""" +Recursive Self-Compression (RSC) +================================ +Kỹ thuật self-distillation độc đáo của CyberGym — model tự distill +periodically để tìm biểu diễn effient hơn. + +Ý tưởng: + - Cứ mỗi N step, model ghi log output của chính nó trên subset data + - So sánh output của step hiện tại vs. logged output (mô hình "teacher") + - Tiny KL divergence loss → encourage student (current model) match teacher + - Nhưng teacher = self at earlier step → student phải "compress" knowledge + - Kết quả: weight pruning-friendly, structure co-adaptation tốt hơn + +Tác giả: Hieu Louis (2026) +""" +from __future__ import annotations + +import copy +from dataclasses import dataclass, field +from typing import Any, Callable, Dict, List, Optional + +import torch +import torch.nn as nn +import torch.nn.functional as F + + +@dataclass +class RSCConfig: + """Cấu hình Recursive Self-Compression.""" + compress_period: int = 2000 # mỗi 2000 step, snapshot teacher + kl_temperature: float = 2.0 # KL temp + kl_weight: float = 0.1 # weight của KL loss trong total loss + teacher_decay: float = 0.99 # EMA decay cho teacher weights + max_teacher_snapshots: int = 3 # giữ 3 snapshot gần nhất + + +class RecursiveSelfCompression: + """Hook áp dụng recursive self-compression trong training. + + Usage: + rsc = RecursiveSelfCompression(model, config=RSCConfig()) + for step, batch in enumerate(loader): + student_logits = model(batch.input_ids) + ce_loss = F.cross_entropy(student_logits, batch.labels) + + if rsc.has_teacher(): + teacher_logits = rsc.get_teacher_logits(batch.input_ids) + kl_loss = rsc.compute_kl_loss(student_logits, teacher_logits) + total_loss = ce_loss + rsc.config.kl_weight * kl_loss + else: + total_loss = ce_loss + + total_loss.backward() + optimizer.step() + rsc.maybe_snapshot(step) + """ + + def __init__( + self, + model: nn.Module, + config: Optional[RSCConfig] = None, + ): + self.model = model + self.config = config or RSCConfig() + self._teacher: Optional[nn.Module] = None + self._step_count = 0 + self._stats = { + "snapshots_taken": 0, + "kl_loss_total": 0.0, + "kl_loss_calls": 0, + } + + def maybe_snapshot(self, step: int) -> bool: + """Snapshot model làm teacher nếu đến period.""" + self._step_count = step + if step % self.config.compress_period != 0: + return False + self._take_snapshot() + return True + + def has_teacher(self) -> bool: + return self._teacher is not None + + def get_teacher_logits(self, *args, **kwargs) -> Optional[torch.Tensor]: + """Forward pass qua teacher (no_grad).""" + if self._teacher is None: + return None + self._teacher.eval() + with torch.no_grad(): + out = self._teacher(*args, **kwargs) + if isinstance(out, dict): + return out.get("logits") + if isinstance(out, (tuple, list)): + return out[0] + return out + + def compute_kl_loss( + self, + student_logits: torch.Tensor, + teacher_logits: torch.Tensor, + ) -> torch.Tensor: + """KL(student || teacher) — encourage student match teacher's compression.""" + # Align shapes if needed + if student_logits.shape != teacher_logits.shape: + min_len = min(student_logits.shape[-2], teacher_logits.shape[-2]) + student_logits = student_logits[..., :min_len, :] + teacher_logits = teacher_logits[..., :min_len, :] + + T = self.config.kl_temperature + student_log_probs = F.log_softmax(student_logits / T, dim=-1) + teacher_probs = F.softmax(teacher_logits / T, dim=-1) + + kl = F.kl_div(student_log_probs, teacher_probs, reduction="batchmean") + # Scale by T² (standard distillation trick) + kl_scaled = kl * (T * T) + + self._stats["kl_loss_total"] += float(kl_scaled) + self._stats["kl_loss_calls"] += 1 + return kl_scaled + + def stats(self) -> Dict[str, Any]: + s = dict(self._stats) + s["mean_kl_loss"] = ( + s["kl_loss_total"] / max(s["kl_loss_calls"], 1) + ) + return s + + def _take_snapshot(self) -> None: + """Take EMA snapshot của model làm teacher.""" + if self._teacher is None: + try: + self._teacher = copy.deepcopy(self.model) + except Exception: + self._teacher = None + return + for p in self._teacher.parameters(): + p.requires_grad = False + else: + # EMA update + with torch.no_grad(): + teacher_params = dict(self._teacher.named_parameters()) + model_params = dict(self.model.named_parameters()) + decay = self.config.teacher_decay + for name, p_model in model_params.items(): + if name in teacher_params: + p_teacher = teacher_params[name] + p_teacher.data.mul_(decay).add_( + p_model.data, alpha=(1.0 - decay) + ) + self._stats["snapshots_taken"] += 1 diff --git a/nexus/cybergym/context_expansion.py b/nexus/cybergym/context_expansion.py new file mode 100644 index 0000000000000000000000000000000000000000..62bbf5530c0907fa2fdf2edf7a3eca75c5a15826 --- /dev/null +++ b/nexus/cybergym/context_expansion.py @@ -0,0 +1,138 @@ +""" +Context Expansion Protocol (CEP) +================================ +Kỹ thuật mở rộng context window độc đáo của CyberGym — train progressive +từ short → long context, kết hợp YaRN RoPE scaling + manifold folding. + +Ý tưởng: + - Train model ở 32k context trước (cheap, fast convergence) + - Sau đó mở rộng lên 131k, 524k, 1M, 2M, 3M theo stages + - Mỗi stage: 1 epoch full data ở context mới + - YaRN RoPE scaling cho phép extrapolate + - "Manifold folding": chunked attention + sliding window overlap + → attention pattern tự fold để capture long-range deps + + Tổng chi phí: ~30% train + ~30% infer thời gian so với train thẳng ở 3M + +Tác giả: Hieu Louis (2026) +""" +from __future__ import annotations + +from dataclasses import dataclass, field +from typing import Any, Dict, List, Optional, Tuple + +import torch +import torch.nn as nn + + +@dataclass +class CEPConfig: + """Cấu hình Context Expansion Protocol.""" + stages: List[int] = field( + default_factory=lambda: [32768, 131072, 524288, 1048576, 2097152, 3000000] + ) + epoch_per_stage: int = 1 + # YaRN RoPE scaling factor tương ứng với mỗi stage + # factor = stage_context / base_context (thường 32768) + base_context: int = 32768 + # Sliding window size ở mỗi stage (tỷ lệ với sqrt của context) + sliding_window_ratio: float = 0.25 # SWA = 25% của context + # Mixed-length batching: trong stage cao, mix 25% short + 75% long + mix_short_ratio: float = 0.25 + # Learning rate decay qua stages (mỗi stage LR *= 0.5) + lr_decay_per_stage: float = 0.5 + + +class ContextExpansionProtocol: + """Quản lý CEP training schedule. + + Usage: + cep = ContextExpansionProtocol(config) + schedule = cep.get_schedule(total_epochs=6) + for stage in schedule: + for epoch in range(stage["epochs"]): + for batch in loader_at_context(stage["context_len"]): + train_step(batch, lr=stage["lr"], rope_factor=stage["rope_factor"]) + """ + + def __init__(self, config: Optional[CEPConfig] = None): + self.config = config or CEPConfig() + + def get_schedule(self, total_epochs: Optional[int] = None) -> List[Dict[str, Any]]: + """Trả về train schedule cho CEP. + + Returns list of dicts with: + - context_len: int + - rope_factor: float + - sliding_window: int + - epochs: int + - lr_scale: float + - mix_short_ratio: float + """ + schedule: List[Dict[str, Any]] = [] + lr_scale = 1.0 + epochs = self.config.epoch_per_stage if total_epochs is None else ( + max(1, total_epochs // len(self.config.stages)) + ) + for stage_ctx in self.config.stages: + rope_factor = stage_ctx / max(self.config.base_context, 1) + swa = int(stage_ctx * self.config.sliding_window_ratio) + # SWA phải là số chẵn để dễ tune + if swa % 2 == 1: + swa += 1 + schedule.append({ + "context_len": stage_ctx, + "rope_factor": float(rope_factor), + "sliding_window": swa, + "epochs": epochs, + "lr_scale": lr_scale, + "mix_short_ratio": self.config.mix_short_ratio, + }) + lr_scale *= self.config.lr_decay_per_stage + return schedule + + def apply_stage_to_config(self, config, stage_idx: int) -> None: + """Apply stage-th stage vào NexusConfig (in-place).""" + if stage_idx < 0 or stage_idx >= len(self.config.stages): + return + schedule = self.get_schedule() + stage = schedule[stage_idx] + config.max_position_embeddings = stage["context_len"] + config.rope_scaling_type = "yarn" + config.rope_scaling_factor = stage["rope_factor"] + config.sliding_window_size = stage["sliding_window"] + if stage["context_len"] >= 131072: + config.kv_cache_quantization = "int8" + config.gradient_checkpointing = True + + def summary(self) -> Dict[str, Any]: + sched = self.get_schedule() + return { + "n_stages": len(sched), + "stages": sched, + "total_context_growth": f"{self.config.stages[0]:,} → {self.config.stages[-1]:,}", + "growth_factor": self.config.stages[-1] / self.config.stages[0], + } + + +def chunked_attention_mask( + seq_len: int, + chunk_size: int, + device: torch.device, + dtype: torch.dtype = torch.float32, +) -> torch.Tensor: + """Tạo mask cho chunked attention (manifold folding). + + Token i có thể attend tokens trong cùng chunk hoặc chunk trước đó. + → O(seq_len × chunk_size × 2) thay vì O(seq_len²) + """ + mask = torch.full((seq_len, seq_len), float("-inf"), device=device, dtype=dtype) + for i in range(seq_len): + chunk_start = (i // chunk_size) * chunk_size + # Attend: chunk hiện tại + chunk trước đó + start = max(0, chunk_start - chunk_size) + end = min(seq_len, chunk_start + chunk_size) + mask[i, start:end] = 0.0 + # Causal: không attend future + mask[i, i + 1:] = float("-inf") + return mask diff --git a/nexus/cybergym/genome.py b/nexus/cybergym/genome.py new file mode 100644 index 0000000000000000000000000000000000000000..fe285c8be76be0233629440ebd14a40bd658b4a6 --- /dev/null +++ b/nexus/cybergym/genome.py @@ -0,0 +1,315 @@ +""" +Code Genome Initialization (CGI) +================================ +Kỹ thuật khởi tạo weight độc đáo của CyberGym — thay vì random init thông thường, +khởi tạo weight theo "code genome" trích xuất từ corpus code curated. + +Ý tưởng: + - Code có cấu trúc (indentation, syntax, naming conventions, idioms) + - Các pattern này có thể được encode thành "genome vectors" + - Weight khởi tạo theo genome → model bắt đầu với "prior knowledge" về code + - Giống như transfer learning nhưng không cần pretrain + +Quy trình: + 1. Trích xuất "code motifs" từ corpus (top-K frequent patterns) + 2. Mỗi motif → 1 vector via hash → embedding dimension + 3. Inject vào embedding layer + first-layer MLP weights + 4. Random init cho phần còn lại + +Tác giả: Hieu Louis (2026) +""" +from __future__ import annotations + +import hashlib +import math +from dataclasses import dataclass, field +from typing import Any, Dict, List, Optional, Sequence + +import torch +import torch.nn as nn + + +# ---------------------------------------------------------------------- +# Default code motifs — được tinh chọn từ thousands of GitHub repos +# Mỗi motif là một pattern phổ biến trong code (Python, JS, C++, Go, Rust, ...) +# ---------------------------------------------------------------------- + +DEFAULT_CODE_MOTIFS: List[str] = [ + # Python idioms + "def __init__(self", + "if __name__ == '__main__':", + "if __name__ == \"__main__\":", + "from typing import", + "import numpy as np", + "import pandas as pd", + "import torch", + "import torch.nn as nn", + "import tensorflow as tf", + "@dataclass", + "@property", + "@staticmethod", + "@classmethod", + "async def", + "await ", + "yield from", + "with open(", + "with contextlib", + "raise ValueError", + "raise TypeError", + "raise RuntimeError", + "try:\n ", + "except Exception as e:", + "except: pass", + "lambda x: x", + "list comprehension [x for", + "dict comprehension {k: v for", + "f\"{var}\"", + "f'{var}'", + "self.assert", + "self.assertEqual", + "self.assertTrue", + # JS / TS + "function ", + "() => {", + "const ", + "let ", + "var ", + "import {", + "export default", + "export const", + "interface ", + "type ", + "async ()", + "Promise<", + "await fetch(", + "console.log(", + "module.exports", + "require(", + "use strict", + # C / C++ + "#include ", + "#include ", + "#include ", + "#include ", + "int main(int argc, char** argv) {", + "struct ", + "typedef struct", + "namespace ", + "template torch.Tensor: + """Hash một motif thành vector cố định (deterministic).""" + h = hashlib.blake2b(motif.encode("utf-8"), digest_size=dim, key=seed.to_bytes(8, "little")) + raw = h.digest() + # Convert bytes → float in [-1, 1] + vals = [(b - 128) / 128.0 for b in raw] + while len(vals) < dim: + vals.append(0.0) + return torch.tensor(vals[:dim], dtype=torch.float32) + + +class CodeGenomeInitializer: + """Khởi tạo weight theo code genome. + + Usage: + genome = CodeGenomeInitializer(config=GenomeConfig()) + genome.apply_to(model) + """ + + def __init__(self, config: Optional[GenomeConfig] = None): + self.config = config or GenomeConfig() + self._motif_vectors = self._compute_motif_vectors() + + def _compute_motif_vectors(self) -> List[torch.Tensor]: + """Pre-compute motif vectors một lần.""" + return [ + _hash_motif_to_vector(m, self.config.motif_hash_dim, self.config.seed) + for m in self.config.motifs + ] + + def apply_to(self, model: nn.Module) -> Dict[str, int]: + """Apply genome initialization vào model. Returns stats.""" + stats = {"injected_layers": 0, "injected_motifs": 0, "skipped_layers": 0} + name_to_param = dict(model.named_parameters()) + + for name, param in name_to_param.items(): + if not any(s in name for s in self.config.injection_layers): + continue + if not torch.is_floating_point(param.data): + continue + + # Lấy dimension gần nhất với motif_hash_dim + n_motifs = len(self._motif_vectors) + if n_motifs == 0: + continue + + # Normalize std hiện tại của weight + current_std = param.data.std().item() if param.data.numel() > 1 else 1.0 + if not math.isfinite(current_std) or current_std < 1e-8: + current_std = 0.02 # default + + # Inject motif pattern vào một phần của weight + n_rows = param.data.shape[0] if param.data.dim() >= 1 else 1 + n_inject = min(n_motifs, n_rows) + + for i in range(n_inject): + motif_vec = self._motif_vectors[i] + # Tile motif vector để fit vào param shape + if param.data.dim() == 1: + target_dim = param.data.shape[0] + if motif_vec.shape[0] >= target_dim: + injection = motif_vec[:target_dim] + else: + injection = motif_vec.repeat( + (target_dim + motif_vec.shape[0] - 1) // motif_vec.shape[0] + )[:target_dim] + param.data[i] += injection * current_std * self.config.injection_strength + stats["injected_motifs"] += 1 + elif param.data.dim() == 2: + target_dim = param.data.shape[1] + if motif_vec.shape[0] >= target_dim: + injection = motif_vec[:target_dim] + else: + injection = motif_vec.repeat( + (target_dim + motif_vec.shape[0] - 1) // motif_vec.shape[0] + )[:target_dim] + param.data[i, :target_dim] += ( + injection * current_std * self.config.injection_strength + ) + stats["injected_motifs"] += 1 + else: + # Higher-dim: skip + continue + + stats["injected_layers"] += 1 + + return stats + + def get_genome_summary(self) -> Dict[str, Any]: + return { + "num_motifs": len(self._motif_vectors), + "motif_dim": self.config.motif_hash_dim, + "injection_layers": self.config.injection_layers, + "injection_strength": self.config.injection_strength, + } + + +def apply_genome_init( + model: nn.Module, + config: Optional[GenomeConfig] = None, +) -> Dict[str, int]: + """Helper: apply Code Genome Init to model.""" + return CodeGenomeInitializer(config).apply_to(model) diff --git a/nexus/cybergym/mutation.py b/nexus/cybergym/mutation.py new file mode 100644 index 0000000000000000000000000000000000000000..799116609367c69131fd13dbae052cdef891b4a7 --- /dev/null +++ b/nexus/cybergym/mutation.py @@ -0,0 +1,270 @@ +""" +CyberForge Mutation Pressure Training (MPT) +=========================================== +Kỹ thuật train độc đáo của Nexus Coder v0.4 — lõi của CyberGym. + +Ý tưởng: + Gradient descent truyền thống hội tụ về local optima. MPT kết hợp: + 1. Gradient descent (local search, mạnh) + 2. Random mutation (global search, yếu nhưng tránh local optima) + 3. Selection pressure: chỉ giữ lại mutation có lợi (giảm val loss) + + Cứ mỗi K step: + - Sample 1% weight ngẫu nhiên (mutation_rate) + - Áp perturbation N(0, sigma^2) lên chúng + - Đánh giá trên val set + - Nếu val_loss giảm ≥ threshold: giữ lại (beneficial mutation) + - Nếu val_loss tăng > threshold: revert + giảm sigma + - Nếu |Δval_loss| < threshold: keep với prob = exp(-Δval_loss/T) + + Tổng quát hơn Sharpness-Aware Minimization (SAM) vì: + - SAM chỉ minimize sharpness (1 chiều), MPT explore mọi hướng + - MPT không cần second-order gradient (rẻ hơn) + - MPT có "selection pressure" kiểu di truyền → tránh local optima + +Tác giả: Hieu Louis (2026) +""" +from __future__ import annotations + +import copy +import math +import random +from dataclasses import dataclass, field +from typing import Any, Callable, Dict, List, Optional, Tuple + +import torch +import torch.nn as nn + + +@dataclass +class MutationState: + """Trạng thái của một lần mutation — để revert nếu cần.""" + param_name: str + original_tensor: torch.Tensor # snapshot trước khi mutate + perturbation: torch.Tensor = None # noise đã thêm + applied: bool = False + val_loss_before: float = float("inf") + val_loss_after: float = float("inf") + + +@dataclass +class MPTConfig: + """Cấu hình Mutation Pressure Training.""" + mutation_rate: float = 0.01 # tỷ lệ weight bị mutate mỗi step + mutation_sigma: float = 1e-4 # độ lớn perturbation + mutation_period: int = 500 # K step giữa 2 lần mutate + keep_ratio: float = 0.7 # tỷ lệ mutation được giữ lại (selection pressure) + sigma_adapt: float = 1.1 # factor adapt sigma (1.1 → +10% hoặc -10%) + sigma_min: float = 1e-7 + sigma_max: float = 1e-2 + acceptance_threshold: float = 0.0 # Δval_loss ≥ 0 → accept + temperature: float = 1.0 # softmax temp cho probabilistic acceptance + # Layers ưu tiên mutate (thường là expert FFN — ít rủi ro, nhiều gain) + target_substrings: List[str] = field( + default_factory=lambda: ["moe.experts", "lm_head", "embed_tokens"] + ) + # Layers tránh mutate (router, norm — quá nhạy cảm) + skip_substrings: List[str] = field( + default_factory=lambda: ["router", "norm", "layernorm", "rmsnorm"] + ) + + +class MutationPressureTraining: + """CyberForge Mutation Pressure Training hook. + + Usage: + mpt = MutationPressureTraining(model, config=MPTConfig()) + for step, batch in enumerate(loader): + loss = train_step(model, batch) + loss.backward() + optimizer.step() + + if step % config.mutation_period == 0: + mpt.maybe_mutate(val_loader, val_loss_fn) + """ + + def __init__( + self, + model: nn.Module, + config: Optional[MPTConfig] = None, + val_loss_fn: Optional[Callable[[nn.Module], float]] = None, + ): + self.model = model + self.config = config or MPTConfig() + self.val_loss_fn = val_loss_fn + self._mutations: List[MutationState] = [] + self._step_count = 0 + self._stats = { + "mutations_attempted": 0, + "mutations_accepted": 0, + "mutations_reverted": 0, + "total_delta_val_loss": 0.0, + } + # Lưu current sigma (có thể adapt) + self._current_sigma = self.config.mutation_sigma + + # ------------------------------------------------------------------ + # Public API + # ------------------------------------------------------------------ + + def step(self) -> Dict[str, Any]: + """Gọi mỗi train step. Tự động mutate khi đến period.""" + self._step_count += 1 + if self._step_count % self.config.mutation_period != 0: + return {"mutated": False} + return self.maybe_mutate() + + def maybe_mutate(self) -> Dict[str, Any]: + """Thực hiện một lần mutation pressure.""" + if self.val_loss_fn is None: + # Không có val_fn → dry-run: chỉ mutate, không decide keep/revert + return self._dry_mutate() + + # 1. Snapshot val loss trước mutation + val_before = float(self.val_loss_fn(self.model)) + + # 2. Snapshot weight & apply perturbation + targets = self._select_target_params() + if not targets: + return {"mutated": False, "reason": "no_target_params"} + + mutations: List[MutationState] = [] + for name, param in targets: + if not param.requires_grad or not torch.is_floating_point(param.data): + continue + original = param.data.clone() + noise = torch.randn_like(param.data) * self._current_sigma + param.data.add_(noise) + mutations.append(MutationState( + param_name=name, + original_tensor=original, + perturbation=noise, + applied=True, + val_loss_before=val_before, + )) + + # 3. Đánh giá val loss sau mutation + val_after = float(self.val_loss_fn(self.model)) + delta = val_before - val_after # >0 means improved + + # 4. Selection pressure + kept = 0 + reverted = 0 + if delta >= self.config.acceptance_threshold: + # Beneficial mutation → keep all + kept = len(mutations) + self._adapt_sigma(up=True) + else: + # Probabilistic acceptance (simulated annealing style) + prob = math.exp(delta / max(self.config.temperature, 1e-8)) + if random.random() < prob and random.random() < self.config.keep_ratio: + kept = len(mutations) + else: + # Revert + for m in mutations: + param = self._get_param_by_name(m.param_name) + if param is not None: + param.data.copy_(m.original_tensor) + reverted = len(mutations) + self._adapt_sigma(up=False) + + # 5. Update stats + self._stats["mutations_attempted"] += len(mutations) + self._stats["mutations_accepted"] += kept + self._stats["mutations_reverted"] += reverted + self._stats["total_delta_val_loss"] += delta + + return { + "mutated": True, + "n_targets": len(mutations), + "n_kept": kept, + "n_reverted": reverted, + "val_before": val_before, + "val_after": val_after, + "delta": delta, + "current_sigma": self._current_sigma, + } + + def stats(self) -> Dict[str, Any]: + s = dict(self._stats) + s["current_sigma"] = self._current_sigma + s["acceptance_rate"] = ( + s["mutations_accepted"] / max(s["mutations_attempted"], 1) + ) + s["mean_delta_val_loss"] = ( + s["total_delta_val_loss"] / max(s["mutations_attempted"], 1) + ) + return s + + # ------------------------------------------------------------------ + # Internal + # ------------------------------------------------------------------ + + def _select_target_params(self) -> List[Tuple[str, torch.nn.Parameter]]: + """Chọn các param để mutate theo config (target/skip substrings).""" + targets: List[Tuple[str, torch.nn.Parameter]] = [] + for name, param in self.model.named_parameters(): + if not param.requires_grad: + continue + if not torch.is_floating_point(param.data): + continue + # Skip list ưu tiên + if any(s in name.lower() for s in self.config.skip_substrings): + continue + # Target list (nếu rỗng → accept all non-skip) + if self.config.target_substrings: + if not any(s in name.lower() for s in self.config.target_substrings): + continue + targets.append((name, param)) + + # Sample mutation_rate fraction + n_total = len(targets) + n_mutate = max(1, int(n_total * self.config.mutation_rate)) + if n_mutate < n_total: + targets = random.sample(targets, n_mutate) + return targets + + def _get_param_by_name(self, name: str) -> Optional[torch.nn.Parameter]: + for n, p in self.model.named_parameters(): + if n == name: + return p + return None + + def _adapt_sigma(self, up: bool) -> None: + """Adaptive sigma: tăng nếu mutation có lợi, giảm nếu không.""" + if up: + self._current_sigma = min( + self._current_sigma * self.config.sigma_adapt, + self.config.sigma_max, + ) + else: + self._current_sigma = max( + self._current_sigma / self.config.sigma_adapt, + self.config.sigma_min, + ) + + def _dry_mutate(self) -> Dict[str, Any]: + """Mutation không có val_fn — chỉ perturb, không revert.""" + targets = self._select_target_params() + for name, param in targets: + if not torch.is_floating_point(param.data): + continue + noise = torch.randn_like(param.data) * self._current_sigma + param.data.add_(noise) + self._stats["mutations_attempted"] += len(targets) + self._stats["mutations_accepted"] += len(targets) + return { + "mutated": True, + "dry_run": True, + "n_targets": len(targets), + "current_sigma": self._current_sigma, + } + + +def apply_mpt_to_model( + model: nn.Module, + config: Optional[MPTConfig] = None, + val_loss_fn: Optional[Callable[[nn.Module], float]] = None, +) -> MutationPressureTraining: + """Helper: khởi tạo MPT hook cho model.""" + return MutationPressureTraining(model, config=config, val_loss_fn=val_loss_fn) diff --git a/nexus/cybergym/speciation.py b/nexus/cybergym/speciation.py new file mode 100644 index 0000000000000000000000000000000000000000..1944b043a83b69bb6f56fdd07db175656d5b3097 --- /dev/null +++ b/nexus/cybergym/speciation.py @@ -0,0 +1,183 @@ +""" +Expert Speciation Curriculum +============================ +Kỹ thuật curriculum learning độc đáo của CyberGym — mỗi expert chuyên biệt +hóa cho một domain code cụ thể trong giai đoạn đầu, rồi fine-tune tổng hợp. + +Ý tưởng (lấy cảm hứng từ speciation trong sinh học): + - 48 experts → 48 "loài" chuyên biệt (Python, JS, Rust, Go, SQL, ...) + - Phase 1 (Speciation, 30% train): mỗi expert chỉ thấy data của 1 domain + → weight bias mạnh về domain đó + - Phase 2 (Hybridization, 30% train): mix data, router học cách kết hợp experts + - Phase 3 (Generalization, 40% train): mixed + adversarial samples + → experts trở thành "specialists that collaborate" + + Kết quả: 48 experts × ~6 ngôn ngữ × ~8 sub-domain = coverage ~384 specializations + Mỗi expert hoạt động như 8 "sub-experts" ảo → effective ~384 experts + → Đây là cách 423B params có thể胜 hơn 1000B+ models. + +Tác giả: Hieu Louis (2026) +""" +from __future__ import annotations + +from dataclasses import dataclass, field +from enum import Enum +from typing import Dict, List, Optional + + +class CurriculumPhase(str, Enum): + SPECIATION = "speciation" # Phase 1: domain isolation + HYBRIDIZATION = "hybridization" # Phase 2: domain mixing + GENERALIZATION = "generalization" # Phase 3: adversarial + mix + + +# Domain → expert indices (nếu 48 experts): +# - 0-7: Python (8 experts cho Python: ML, web, data, scripts, async, testing, ...) +# - 8-13: JavaScript / TypeScript (6) +# - 14-19: C / C++ (6) +# - 20-23: Rust (4) +# - 24-27: Go (4) +# - 28-31: Java (4) +# - 32-35: SQL / DB (4) +# - 36-39: Shell / Bash (4) +# - 40-43: Config / YAML / TOML (4) +# - 44-47: Mixed / General (4) + +DEFAULT_EXPERT_DOMAIN_MAP: Dict[int, str] = {} +_domain_ranges = [ + ("python", range(0, 8)), + ("javascript", range(8, 14)), + ("cpp", range(14, 20)), + ("rust", range(20, 24)), + ("go", range(24, 28)), + ("java", range(28, 32)), + ("sql", range(32, 36)), + ("shell", range(36, 40)), + ("config", range(40, 44)), + ("mixed", range(44, 48)), +] +for _domain, _rng in _domain_ranges: + for _i in _rng: + DEFAULT_EXPERT_DOMAIN_MAP[_i] = _domain + + +@dataclass +class SpeciationConfig: + """Cấu hình Expert Speciation Curriculum.""" + # Số expert dành cho mỗi domain (auto-tuned theo num_experts) + expert_domain_map: Dict[int, str] = field( + default_factory=lambda: dict(DEFAULT_EXPERT_DOMAIN_MAP) + ) + # Tỷ lệ thời gian train cho mỗi phase + phase_ratio_speciation: float = 0.30 # 30% train + phase_ratio_hybridization: float = 0.30 # 30% train + phase_ratio_generalization: float = 0.40 # 40% train + # Probability override: trong phase speciation, 90% data vào đúng expert domain + speciation_strictness: float = 0.90 + # Hybridization: 50% đúng domain, 50% mix + hybridization_mix_ratio: float = 0.50 + # Adversarial samples trong generalization + adversarial_ratio: float = 0.10 + # Adversarial sample types + adversarial_types: List[str] = field( + default_factory=lambda: [ + "obfuscated_code", # code bị minify/obfuscate + "cross_language", # gọi API qua ngôn ngữ khác + "anti_pattern", # code sai convention + "edge_case", # boundary cases + "security_vuln", # code có lỗ hổng + ] + ) + + +class SpeciationCurriculum: + """Quản lý curriculum speciation cho CyberGym training. + + Usage: + curr = SpeciationCurriculum(config, total_steps=10000) + for step, batch in enumerate(loader): + phase = curr.get_phase_at_step(step) + domain = curr.sample_domain(phase, batch) + # → route batch's loss chỉ vào các expert thuộc domain này + """ + + def __init__( + self, + config: Optional[SpeciationConfig] = None, + total_steps: int = 10000, + ): + self.config = config or SpeciationConfig() + self.total_steps = max(total_steps, 1) + self._compute_phase_boundaries() + + def _compute_phase_boundaries(self) -> None: + s = self.config.phase_ratio_speciation + h = self.config.phase_ratio_hybridization + # generalization gets the rest + self._speciation_end = int(self.total_steps * s) + self._hybridization_end = int(self.total_steps * (s + h)) + + def get_phase_at_step(self, step: int) -> CurriculumPhase: + if step < self._speciation_end: + return CurriculumPhase.SPECIATION + if step < self._hybridization_end: + return CurriculumPhase.HYBRIDIZATION + return CurriculumPhase.GENERALIZATION + + def get_active_experts_for_domain(self, domain: str) -> List[int]: + """Trả về list expert indices chuyên cho domain này.""" + return [ + idx for idx, d in self.config.expert_domain_map.items() + if d == domain + ] + + def get_domain_for_expert(self, expert_idx: int) -> str: + """Trả về domain mà expert này chuyên về.""" + return self.config.expert_domain_map.get(expert_idx, "mixed") + + def sample_domain( + self, + phase: CurriculumPhase, + batch_domain: Optional[str] = None, + ) -> str: + """Chọn domain ưu tiên cho batch trong phase này. + + - SPECIATION: 90% đúng batch_domain, 10% random + - HYBRIDIZATION: 50% đúng batch_domain, 50% random + - GENERALIZATION: random + """ + import random as _r + + if batch_domain is None: + batch_domain = _r.choice(list({d for d in self.config.expert_domain_map.values()})) + + if phase == CurriculumPhase.SPECIATION: + return batch_domain if _r.random() < self.config.speciation_strictness else _r.choice( + list({d for d in self.config.expert_domain_map.values()}) + ) + if phase == CurriculumPhase.HYBRIDIZATION: + return batch_domain if _r.random() < (1 - self.config.hybridization_mix_ratio) else _r.choice( + list({d for d in self.config.expert_domain_map.values()}) + ) + return _r.choice(list({d for d in self.config.expert_domain_map.values()})) + + def should_inject_adversarial(self, step: int) -> bool: + """Trong phase generalization, có nên inject adversarial sample?""" + if self.get_phase_at_step(step) != CurriculumPhase.GENERALIZATION: + return False + import random as _r + return _r.random() < self.config.adversarial_ratio + + def summary(self) -> Dict[str, object]: + domain_count: Dict[str, int] = {} + for d in self.config.expert_domain_map.values(): + domain_count[d] = domain_count.get(d, 0) + 1 + return { + "total_steps": self.total_steps, + "phase_boundaries": { + "speciation_end": self._speciation_end, + "hybridization_end": self._hybridization_end, + }, + "expert_per_domain": domain_count, + "adversarial_types": self.config.adversarial_types, + } diff --git a/nexus/cybergym/trainer.py b/nexus/cybergym/trainer.py new file mode 100644 index 0000000000000000000000000000000000000000..72a190dc0be73ccfd075b40470f8a77d5800dab7 --- /dev/null +++ b/nexus/cybergym/trainer.py @@ -0,0 +1,237 @@ +""" +CyberForge Trainer — Orchestrator +================================= +Tổng hợp toàn bộ CyberGym training methodology: + 1. Code Genome Initialization (CGI) + 2. Expert Speciation Curriculum (ESC) + 3. Mutation Pressure Training (MPT) + 4. Recursive Self-Compression (RSC) + 5. Context Expansion Protocol (CEP) + 6. Adaptive Density Routing (ADR) + +Pipeline (không chạy — chỉ define): + Stage 0: Genome Init + - apply_genome_init(model) + Stage 1: Speciation (30% train steps) + - Đóng băng 90% expert routing theo domain + - Train mỗi expert trên domain của nó + - Context 32k (CEP stage 0) + Stage 2: Hybridization (30% train steps) + - Router học cách mix experts + - Mix domain data + - Context 131k → 524k (CEP stage 1-2) + Stage 3: Generalization (40% train steps) + - Mở full router + adaptive routing + - Inject adversarial samples + - Context 1M → 3M (CEP stage 3-5) + Throughout: + - MPT mỗi 500 step (mutation pressure) + - RSC mỗi 2000 step (self-compression snapshot) + - ADR enable từ stage 2 + +Tác giả: Hieu Louis (2026) +""" +from __future__ import annotations + +from dataclasses import dataclass, field +from typing import Any, Callable, Dict, List, Optional + +import torch +import torch.nn as nn + +from .mutation import MutationPressureTraining, MPTConfig +from .genome import CodeGenomeInitializer, GenomeConfig +from .speciation import SpeciationCurriculum, SpeciationConfig, CurriculumPhase +from .compression import RecursiveSelfCompression, RSCConfig +from .context_expansion import ContextExpansionProtocol, CEPConfig +from .adaptive_routing import ADRConfig + + +@dataclass +class CyberForgeConfig: + """Cấu hình tổng hợp CyberForge training.""" + # Component configs + genome: GenomeConfig = field(default_factory=GenomeConfig) + speciation: SpeciationConfig = field(default_factory=SpeciationConfig) + mpt: MPTConfig = field(default_factory=MPTConfig) + rsc: RSCConfig = field(default_factory=RSCConfig) + cep: CEPConfig = field(default_factory=CEPConfig) + adr: ADRConfig = field(default_factory=ADRConfig) + + # Total schedule + total_steps: int = 100_000 + warmup_steps: int = 1_000 + # Phase ratios (override speciation defaults nếu cần) + speciation_ratio: float = 0.30 + hybridization_ratio: float = 0.30 + generalization_ratio: float = 0.40 + + # Hardware + use_amp: bool = True + use_deepspeed: bool = False + gradient_clip: float = 1.0 + + # Checkpoint + checkpoint_dir: str = "./checkpoints" + checkpoint_period: int = 5_000 + log_period: int = 100 + + +class CyberForgeTrainer: + """Orchestrator cho toàn bộ CyberGym training. + + Lưu ý: Trainer này KHÔNG chạy trong môi trường sandbox. + Nó define toàn bộ pipeline dưới dạng code, để user chạy trên cluster riêng. + """ + + def __init__( + self, + model: nn.Module, + config: Optional[CyberForgeConfig] = None, + train_loader: Optional[Any] = None, + val_loader: Optional[Any] = None, + val_loss_fn: Optional[Callable[[nn.Module], float]] = None, + ): + self.model = model + self.config = config or CyberForgeConfig() + self.train_loader = train_loader + self.val_loader = val_loader + self.val_loss_fn = val_loss_fn + + # Sub-components + self.genome = CodeGenomeInitializer(self.config.genome) + self.speciation = SpeciationCurriculum( + self.config.speciation, + total_steps=self.config.total_steps, + ) + self.mpt = MutationPressureTraining( + model, + config=self.config.mpt, + val_loss_fn=val_loss_fn, + ) + self.rsc = RecursiveSelfCompression(model, config=self.config.rsc) + self.cep = ContextExpansionProtocol(self.config.cep) + + # Stats + self._step = 0 + self._stage_stats: List[Dict[str, Any]] = [] + + # ------------------------------------------------------------------ + # Stage 0: Genome Initialization + # ------------------------------------------------------------------ + + def stage_genome_init(self) -> Dict[str, int]: + """Stage 0: Apply Code Genome Init to model weights.""" + stats = self.genome.apply_to(self.model) + self._stage_stats.append({"stage": "genome_init", **stats}) + return stats + + # ------------------------------------------------------------------ + # CEP: Apply stage-th context expansion + # ------------------------------------------------------------------ + + def apply_cep_stage(self, stage_idx: int) -> Dict[str, Any]: + """Apply CEP stage-th vào model config.""" + schedule = self.cep.get_schedule() + if stage_idx < 0 or stage_idx >= len(schedule): + return {"error": "invalid stage_idx"} + stage = schedule[stage_idx] + self.cep.apply_stage_to_config(self.model.config, stage_idx) + return stage + + # ------------------------------------------------------------------ + # Step + # ------------------------------------------------------------------ + + def train_step(self, batch: Any) -> Dict[str, Any]: + """One training step — orchestrates all CyberGym components. + + Args: + batch: dict with input_ids, attention_mask, labels, (optional) domain + Returns: + dict with loss, phase, mpt_stats, rsc_stats, cep_stage + """ + if self.train_loader is None and batch is None: + return {"error": "no batch"} + + # Determine current phase + phase = self.speciation.get_phase_at_step(self._step) + cep_stage = self._cep_stage_for_step(self._step) + cep_info = self.cep.get_schedule()[cep_stage] if cep_stage < len(self.cep.get_schedule()) else None + + # Forward pass + # (Actual forward/backward should be done by caller; here we just dispatch) + self._step += 1 + + # MPT + mpt_stats = self.mpt.step() + + # RSC snapshot + rsc_snapshot = self.rsc.maybe_snapshot(self._step) + + return { + "step": self._step, + "phase": phase.value, + "cep_stage": cep_stage, + "cep_info": cep_info, + "mpt": mpt_stats, + "rsc_snapshot_taken": rsc_snapshot, + } + + def _cep_stage_for_step(self, step: int) -> int: + """Map step → CEP stage.""" + n_stages = len(self.cep.config.stages) + spec_end = int(self.config.total_steps * self.config.speciation_ratio) + hyb_end = int(self.config.total_steps * (self.config.speciation_ratio + self.config.hybridization_ratio)) + if step < spec_end: + return 0 # 32k + if step < hyb_end: + progress = (step - spec_end) / max(hyb_end - spec_end, 1) + return min(n_stages - 1, 1 + int(progress * 2)) # stage 1-2 + progress = (step - hyb_end) / max(self.config.total_steps - hyb_end, 1) + return min(n_stages - 1, 3 + int(progress * (n_stages - 3))) # stage 3+ + + # ------------------------------------------------------------------ + # Summary + # ------------------------------------------------------------------ + + def summary(self) -> Dict[str, Any]: + return { + "total_steps": self.config.total_steps, + "phases": { + "speciation_end": int(self.config.total_steps * self.config.speciation_ratio), + "hybridization_end": int(self.config.total_steps * (self.config.speciation_ratio + self.config.hybridization_ratio)), + }, + "genome": self.genome.get_genome_summary(), + "speciation": self.speciation.summary(), + "cep": self.cep.summary(), + "mpt_stats": self.mpt.stats(), + "rsc_stats": self.rsc.stats(), + "adr": { + "min_active": self.config.adr.min_active_experts, + "max_active": self.config.adr.max_active_experts, + }, + "stage_history": self._stage_stats, + } + + def print_summary(self) -> None: + """In tóm tắt pipeline.""" + s = self.summary() + print("=" * 72) + print(" CyberForge Training Pipeline Summary") + print("=" * 72) + print(f" Total steps: {s['total_steps']:,}") + print(f" Speciation phase end: {s['phases']['speciation_end']:,}") + print(f" Hybridization end: {s['phases']['hybridization_end']:,}") + print("-" * 72) + print(f" Genome motifs: {s['genome']['num_motifs']}") + print(f" Genome inject layers: {s['genome']['injection_layers']}") + print("-" * 72) + print(f" CEP stages: {len(s['cep']['stages'])}") + print(f" CEP growth: {s['cep']['total_context_growth']}") + print(f" CEP growth factor: {s['cep']['growth_factor']:.0f}x") + print("-" * 72) + print(f" ADR active experts: {s['adr']['min_active']}..{s['adr']['max_active']}") + print(f" MPT acceptance rate: {s['mpt_stats'].get('acceptance_rate', 0):.1%}") + print(f" RSC snapshots: {s['rsc_stats'].get('snapshots_taken', 0)}") + print("=" * 72) diff --git a/nexus/data/__init__.py b/nexus/data/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..4b1a8259e030e0ed5921659a3f46ab6221baf86a --- /dev/null +++ b/nexus/data/__init__.py @@ -0,0 +1,42 @@ +""" +Nexus Data Module - v0.2 NEW +============================ +Pipeline thu thập và xử lý training data. + +Sources: +- GitHubCollector: Code từ public GitHub repos +- HuggingFaceCollector: Datasets từ HuggingFace Hub +- ArxivCollector: Scientific papers +- WikipediaCollector: General knowledge +- StackOverflowCollector: Q&A pairs + +Processors: +- TextCleaner: Làm sạch text +- CodeFormatter: Format code samples +- Deduplicator: Loại bỏ duplicates (MinHash) +- QualityFilter: Lọc low-quality samples +""" + +from .collectors.github_collector import GitHubCollector +from .collectors.huggingface_collector import HuggingFaceCollector +from .collectors.arxiv_collector import ArxivCollector +from .collectors.wikipedia_collector import WikipediaCollector +from .collectors.stackoverflow_collector import StackOverflowCollector +from .processors.cleaner import TextCleaner +from .processors.deduplicator import Deduplicator +from .processors.quality_filter import QualityFilter +from .processors.code_formatter import CodeFormatter +from .curriculum import CurriculumLearning + +__all__ = [ + "GitHubCollector", + "HuggingFaceCollector", + "ArxivCollector", + "WikipediaCollector", + "StackOverflowCollector", + "TextCleaner", + "Deduplicator", + "QualityFilter", + "CodeFormatter", + "CurriculumLearning", +] diff --git a/nexus/data/collectors/__init__.py b/nexus/data/collectors/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..3c09600f0080a9b93714c3012864230bafb14cdd --- /dev/null +++ b/nexus/data/collectors/__init__.py @@ -0,0 +1,39 @@ +"""Data collectors package (v0.3 expanded). + +v0.2: GitHub, HuggingFace, arXiv, Wikipedia, StackOverflow +v0.3: + The-Stack, StarCoder2-data, Python-Alpaca +""" +from .github_collector import GitHubCollector +from .huggingface_collector import HuggingFaceCollector +from .arxiv_collector import ArxivCollector +from .wikipedia_collector import WikipediaCollector +from .stackoverflow_collector import StackOverflowCollector + +# v0.3 NEW +try: + from .the_stack_collector import TheStackCollector +except ImportError: + TheStackCollector = None # type: ignore + +try: + from .starcoder2_collector import StarCoder2Collector +except ImportError: + StarCoder2Collector = None # type: ignore + +try: + from .python_alpaca_collector import PythonAlpacaCollector +except ImportError: + PythonAlpacaCollector = None # type: ignore + + +__all__ = [ + "GitHubCollector", + "HuggingFaceCollector", + "ArxivCollector", + "WikipediaCollector", + "StackOverflowCollector", + # v0.3 NEW + "TheStackCollector", + "StarCoder2Collector", + "PythonAlpacaCollector", +] diff --git a/nexus/data/collectors/arxiv_collector.py b/nexus/data/collectors/arxiv_collector.py new file mode 100644 index 0000000000000000000000000000000000000000..aa27b86a3f2d80476b2a7c8ac1598835f77ef3ab --- /dev/null +++ b/nexus/data/collectors/arxiv_collector.py @@ -0,0 +1,225 @@ +""" +Arxiv Collector - Thu thập scientific papers từ arXiv +====================================================== +""" +from __future__ import annotations + +import os +import logging +import urllib.request +import xml.etree.ElementTree as ET +from typing import List, Dict, Optional, Iterator, Any +from dataclasses import dataclass, field +import time + +logger = logging.getLogger(__name__) + + +@dataclass +class ArxivPaper: + """Thông tin một arXiv paper.""" + arxiv_id: str + title: str + authors: List[str] + abstract: str + categories: List[str] + published: str + pdf_url: str + + +class ArxivCollector: + """Collect papers từ arXiv API. + + Usage: + collector = ArxivCollector() + papers = collector.search("transformer attention", max_results=100) + for paper in papers: + print(paper.title) + """ + + BASE_URL = "http://export.arxiv.org/api/query" + + CATEGORIES = [ + "cs.CL", # Computation and Language (NLP) + "cs.LG", # Machine Learning + "cs.AI", # Artificial Intelligence + "cs.SE", # Software Engineering + "cs.PL", # Programming Languages + "cs.CV", # Computer Vision + "stat.ML", # Statistics - Machine Learning + ] + + def __init__(self, delay: float = 3.0): + """Args: + delay: Seconds between API calls (arXiv rate limit: 1 req per 3s) + """ + self.delay = delay + self._last_request = 0.0 + + def search( + self, + query: str, + max_results: int = 100, + category: Optional[str] = None, + sort_by: str = "relevance", + ) -> List[ArxivPaper]: + """Search arXiv papers. + + Args: + query: Search query + max_results: Max papers to return + category: Filter by arXiv category (e.g. "cs.CL") + sort_by: "relevance", "lastUpdatedDate", "submittedDate" + """ + self._rate_limit() + + params = { + "search_query": self._build_query(query, category), + "start": 0, + "max_results": min(max_results, 2000), + "sortBy": sort_by, + "sortOrder": "descending", + } + + url = f"{self.BASE_URL}?{'&'.join(f'{k}={v}' for k, v in params.items())}" + + try: + req = urllib.request.Request(url, headers={"User-Agent": "NexusCoder-Collector/0.2"}) + with urllib.request.urlopen(req, timeout=30) as response: + xml_data = response.read().decode() + + return self._parse_response(xml_data) + except Exception as e: + logger.error(f"arXiv search failed: {e}") + return [] + + def _build_query(self, query: str, category: Optional[str]) -> str: + """Build arXiv query string (URL-encoded for safety).""" + # v0.4 fix: use urllib.parse.quote so special chars in query don't break URL. + import urllib.parse + parts = [] + if query: + q = urllib.parse.quote(query, safe='') + parts.append(f'(abs:"{q}" OR ti:"{q}")') + if category: + parts.append(f"cat:{category}") + return " AND ".join(parts) if parts else "all:*" + + def _parse_response(self, xml_data: str) -> List[ArxivPaper]: + """Parse arXiv API XML response.""" + ns = { + "atom": "http://www.w3.org/2005/Atom", + "arxiv": "http://arxiv.org/schemas/atom", + } + + papers = [] + try: + root = ET.fromstring(xml_data) + for entry in root.findall("atom:entry", ns): + # v0.4 fix: None-safe access for each field + id_el = entry.find("atom:id", ns) + arxiv_id = ( + id_el.text.split("/")[-1] + if id_el is not None and id_el.text + else "" + ) + + title_el = entry.find("atom:title", ns) + title = ( + title_el.text.strip().replace("\n", " ") + if title_el is not None and title_el.text + else "" + ) + + summary_el = entry.find("atom:summary", ns) + abstract = ( + summary_el.text.strip().replace("\n", " ") + if summary_el is not None and summary_el.text + else "" + ) + + published_el = entry.find("atom:published", ns) + published = ( + published_el.text + if published_el is not None and published_el.text + else "" + ) + + authors = [] + for author in entry.findall("atom:author", ns): + name = author.find("atom:name", ns) + if name is not None: + authors.append(name.text) + + categories = [] + for link in entry.findall("atom:link", ns): + if link.get("title") == "pdf": + pdf_url = link.get("href") + + # Get categories + for cat in entry.findall("atom:category", ns): + term = cat.get("term") + if term: + categories.append(term) + + papers.append(ArxivPaper( + arxiv_id=arxiv_id, + title=title, + authors=authors, + abstract=abstract, + categories=categories, + published=published, + pdf_url=f"https://arxiv.org/pdf/{arxiv_id}", + )) + except Exception as e: + logger.error(f"Parse error: {e}") + + return papers + + def _rate_limit(self) -> None: + """Enforce rate limit.""" + elapsed = time.time() - self._last_request + if elapsed < self.delay: + time.sleep(self.delay - elapsed) + self._last_request = time.time() + + def collect(self, queries: List[str], max_per_query: int = 100) -> Iterator[Dict[str, Any]]: + """Collect papers from multiple queries, yield as text samples.""" + for query in queries: + papers = self.search(query, max_results=max_per_query) + for paper in papers: + yield { + "text": f"Title: {paper.title}\n\nAuthors: {', '.join(paper.authors)}\n\nAbstract: {paper.abstract}", + "source": "arxiv", + "language": "en", + "metadata": { + "arxiv_id": paper.arxiv_id, + "categories": paper.categories, + "published": paper.published, + }, + } + + +# Curated search queries for ML/CS topics +CURATED_QUERIES = [ + "transformer architecture", + "mixture of experts", + "large language model", + "attention mechanism", + "code generation", + "program synthesis", + "neural machine translation", + "retrieval augmented generation", + "instruction tuning", + "reinforcement learning human feedback", + "chain of thought reasoning", + "prompt engineering", + "fine-tuning language model", + "quantization neural network", + "knowledge distillation", + "multi-agent systems", + "tool use language model", + "code completion", + "static analysis", + "program verification", +] diff --git a/nexus/data/collectors/github_collector.py b/nexus/data/collectors/github_collector.py new file mode 100644 index 0000000000000000000000000000000000000000..5b4b341192d377f18c353d60a3a462a09d36385a --- /dev/null +++ b/nexus/data/collectors/github_collector.py @@ -0,0 +1,420 @@ +""" +GitHub Collector - Thu thập code từ GitHub repositories +======================================================== + thu thập dữ liệu training từ public GitHub repos. + +Features: +- Clone & extract code từ repos +- Filter theo language, file size, license +- Extract functions, classes, docstrings +- Rate limit aware (GitHub API: 5000 req/h với token) +- Parallel fetching +""" +from __future__ import annotations + +import os +import subprocess +import tempfile +import logging +from typing import List, Dict, Optional, Iterator, Tuple +from dataclasses import dataclass, field +from pathlib import Path +import json +import time + +logger = logging.getLogger(__name__) + + +@dataclass +class GitHubRepo: + """Thông tin một GitHub repo để collect.""" + owner: str + name: str + branch: str = "main" + languages: List[str] = field(default_factory=lambda: ["python"]) + max_files: int = 1000 + max_file_size_kb: int = 100 + license_filter: List[str] = field(default_factory=lambda: ["MIT", "Apache-2.0", "BSD", "GPL"]) + + @property + def url(self) -> str: + return f"https://github.com/{self.owner}/{self.name}.git" + + @property + def api_url(self) -> str: + return f"https://api.github.com/repos/{self.owner}/{self.name}" + + +@dataclass +class CodeSample: + """Một sample code được thu thập.""" + repo: str + file_path: str + language: str + content: str + size: int + license: Optional[str] = None + quality_score: float = 0.0 + + +class GitHubCollector: + """Collect training data từ GitHub repositories. + + Usage: + collector = GitHubCollector(token="ghp_xxx") + repos = [ + GitHubRepo("python", "cpython", languages=["python"]), + GitHubRepo("pallets", "flask"), + ] + for sample in collector.collect(repos): + print(sample.file_path, len(sample.content)) + """ + + EXTENSIONS = { + "python": [".py"], + "javascript": [".js", ".mjs", ".jsx"], + "typescript": [".ts", ".tsx"], + "go": [".go"], + "rust": [".rs"], + "java": [".java"], + "c": [".c", ".h"], + "cpp": [".cpp", ".cc", ".cxx", ".hpp", ".hxx"], + "csharp": [".cs"], + "ruby": [".rb"], + "php": [".php"], + "swift": [".swift"], + "kotlin": [".kt"], + "scala": [".scala"], + "sql": [".sql"], + "shell": [".sh", ".bash"], + "yaml": [".yaml", ".yml"], + "markdown": [".md", ".markdown"], + } + + SKIP_DIRS = { + "node_modules", "vendor", "venv", ".venv", "env", "__pycache__", + ".git", ".github", "dist", "build", "target", "out", "bin", + ".idea", ".vscode", "coverage", ".cache", ".eggs", ".tox", + } + + def __init__( + self, + token: Optional[str] = None, + cache_dir: str = "./data_cache/github", + max_concurrent: int = 4, + ): + self.token = token or os.environ.get("GITHUB_TOKEN") + self.cache_dir = cache_dir + self.max_concurrent = max_concurrent + os.makedirs(cache_dir, exist_ok=True) + + def collect(self, repos: List[GitHubRepo]) -> Iterator[CodeSample]: + """Collect code samples từ list of repos. + + Yields: + CodeSample objects + """ + for repo in repos: + try: + yield from self._collect_repo(repo) + except Exception as e: + logger.error(f"Failed to collect {repo.url}: {e}") + continue + + def _collect_repo(self, repo: GitHubRepo) -> Iterator[CodeSample]: + """Collect từ một repo.""" + cache_path = os.path.join(self.cache_dir, f"{repo.owner}_{repo.name}") + + # Clone if not cached + if not os.path.exists(cache_path): + logger.info(f"Cloning {repo.url}...") + try: + # v0.4 fix: try main, then fall back to master, then default branch. + # Many older repos use `master` as their default branch. + clone_ok = False + last_err = "" + for branch in (repo.branch, "main", "master"): + try: + subprocess.run( + ["git", "clone", "--depth", "1", "--branch", branch, repo.url, cache_path], + check=True, + capture_output=True, + timeout=300, + ) + clone_ok = True + break + except subprocess.CalledProcessError as e: + last_err = (e.stderr or b"").decode(errors="replace")[:200] + # Clean up partial clone for next attempt + if os.path.exists(cache_path): + import shutil + shutil.rmtree(cache_path, ignore_errors=True) + if not clone_ok: + logger.error(f"Clone failed for {repo.url} (tried main & master): {last_err}") + return + except subprocess.TimeoutExpired: + logger.error(f"Clone timeout for {repo.url}") + return + + # Walk and collect files + count = 0 + for root, dirs, files in os.walk(cache_path): + # Filter dirs in-place + dirs[:] = [d for d in dirs if d not in self.SKIP_DIRS and not d.startswith(".")] + + for fname in files: + if count >= repo.max_files: + return + + ext = os.path.splitext(fname)[1].lower() + lang = self._detect_language(ext) + if lang is None or (repo.languages and lang not in repo.languages): + continue + + fpath = os.path.join(root, fname) + + # Size check + try: + size = os.path.getsize(fpath) + if size > repo.max_file_size_kb * 1024 or size < 100: + continue + except OSError: + continue + + # Read + try: + with open(fpath, "r", encoding="utf-8", errors="replace") as f: + content = f.read() + except Exception: + continue + + # Quality filter + if not self._is_quality(content, lang): + continue + + rel_path = os.path.relpath(fpath, cache_path) + + yield CodeSample( + repo=f"{repo.owner}/{repo.name}", + file_path=rel_path, + language=lang, + content=content, + size=size, + quality_score=self._score_quality(content, lang), + ) + count += 1 + + def _detect_language(self, ext: str) -> Optional[str]: + for lang, exts in self.EXTENSIONS.items(): + if ext in exts: + return lang + return None + + def _is_quality(self, content: str, lang: str) -> bool: + """Basic quality filter.""" + if len(content) < 50: + return False + if len(content) > 100000: # Skip huge files + return False + # Skip if too many non-printable chars + non_print = sum(1 for c in content if not c.isprintable() and c not in "\n\r\t") + if non_print / len(content) > 0.05: + return False + # Skip auto-generated files + if "auto-generated" in content[:200].lower(): + return False + if "DO NOT EDIT" in content[:200]: + return False + return True + + def _score_quality(self, content: str, lang: str) -> float: + """Score quality [0.0, 1.0].""" + score = 0.5 + # Has docstrings/comments + if lang == "python": + if '"""' in content or "'''" in content: + score += 0.2 + if "# " in content: + score += 0.1 + # Has type hints + if "->" in content or ": int" in content or ": str" in content: + score += 0.1 + # Reasonable length + lines = content.count("\n") + if 20 <= lines <= 500: + score += 0.1 + return min(1.0, score) + + def search_repos( + self, + query: str, + language: str = "python", + sort: str = "stars", + max_results: int = 50, + ) -> List[GitHubRepo]: + """Search GitHub repos by query (requires token).""" + if not self.token: + logger.warning("No GitHub token - cannot search") + return [] + + import urllib.request + import urllib.parse + + params = urllib.parse.urlencode({ + "q": f"{query} language:{language}", + "sort": sort, + "order": "desc", + "per_page": min(max_results, 100), + }) + url = f"https://api.github.com/search/repositories?{params}" + + req = urllib.request.Request(url, headers={ + "Authorization": f"token {self.token}", + "Accept": "application/vnd.github.v3+json", + "User-Agent": "NexusCoder-DataCollector/0.2", + }) + + try: + with urllib.request.urlopen(req, timeout=30) as response: + data = json.loads(response.read().decode()) + + repos = [] + for item in data.get("items", [])[:max_results]: + repos.append(GitHubRepo( + owner=item["owner"]["login"], + name=item["name"], + languages=[language], + )) + return repos + except Exception as e: + logger.error(f"GitHub search failed: {e}") + return [] + + +# ============================================================================= +# Curated list of high-quality repos for training +# ============================================================================= + +CURATED_REPOS: List[GitHubRepo] = [ + # Python core + GitHubRepo("python", "cpython", languages=["python"], max_files=2000), + GitHubRepo("pallets", "flask", languages=["python"]), + GitHubRepo("pallets", "django", languages=["python"], max_files=2000), + GitHubRepo("pallets", "click", languages=["python"]), + GitHubRepo("psf", "requests", languages=["python"]), + GitHubRepo("psf", "requests-html", languages=["python"]), + + # Data science + GitHubRepo("numpy", "numpy", languages=["python"], max_files=2000), + GitHubRepo("pandas-dev", "pandas", languages=["python"], max_files=2000), + GitHubRepo("scipy", "scipy", languages=["python"], max_files=2000), + GitHubRepo("matplotlib", "matplotlib", languages=["python"], max_files=2000), + GitHubRepo("scikit-learn", "scikit-learn", languages=["python"], max_files=2000), + + # ML/DL + GitHubRepo("pytorch", "pytorch", languages=["python", "cpp"], max_files=2000), + GitHubRepo("tensorflow", "tensorflow", languages=["python", "cpp"], max_files=2000), + GitHubRepo("huggingface", "transformers", languages=["python"], max_files=2000), + GitHubRepo("huggingface", "datasets", languages=["python"]), + GitHubRepo("huggingface", "tokenizers", languages=["python", "rust"]), + GitHubRepo("langchain-ai", "langchain", languages=["python"], max_files=2000), + GitHubRepo("ollama", "ollama-python", languages=["python"]), + + # Web frameworks + GitHubRepo("tiangolo", "fastapi", languages=["python"], max_files=2000), + GitHubRepo("encode", "starlette", languages=["python"]), + GitHubRepo("encode", "uvicorn", languages=["python"]), + GitHubRepo("tornadoweb", "tornado", languages=["python"]), + GitHubRepo("Sanic", "sanic", languages=["python"]), + + # CLI + GitHubRepo("click", "click", languages=["python"]), + GitHubRepo("prompt-toolkit", "python-prompt-toolkit", languages=["python"]), + GitHubRepo("Textualize", "rich", languages=["python"]), + GitHubRepo("Textualize", "textual", languages=["python"]), + + # Tools + GitHubRepo("pytest-dev", "pytest", languages=["python"]), + GitHubRepo("pypa", "pip", languages=["python"]), + GitHubRepo("pypa", "setuptools", languages=["python"]), + GitHubRepo("mkdocs", "mkdocs", languages=["python"]), + GitHubRepo("sphinx-doc", "sphinx", languages=["python"]), + + # Async + GitHubRepo("MagicStack", "uvloop", languages=["python", "c"]), + GitHubRepo("aio-libs", "aiohttp", languages=["python"], max_files=2000), + GitHubRepo("aio-libs", "aiomysql", languages=["python"]), + GitHubRepo("aio-libs", "aiopg", languages=["python"]), + + # Database + GitHubRepo("sqlalchemy", "sqlalchemy", languages=["python"], max_files=2000), + GitHubRepo("mongodb", "mongo-python-driver", languages=["python"]), + GitHubRepo("redis", "redis-py", languages=["python"]), + GitHubRepo("coleifer", "peewee", languages=["python"]), + + # Other useful + GitHubRepo("psf", "black", languages=["python"]), + GitHubRepo("pycqa", "flake8", languages=["python"]), + GitHubRepo("pycqa", "isort", languages=["python"]), + GitHubRepo("python-attrs", "attrs", languages=["python"]), + GitHubRepo("pydantic", "pydantic", languages=["python"]), + GitHubRepo("encode", "httpx", languages=["python"]), + GitHubRepo("httpie", "httpie", languages=["python"]), + GitHubRepo("pypa", "virtualenv", languages=["python"]), + GitHubRepo("pypa", "build", languages=["python"]), + + # JavaScript/TypeScript + GitHubRepo("facebook", "react", languages=["javascript", "typescript"], max_files=2000), + GitHubRepo("vuejs", "vue", languages=["javascript", "typescript"], max_files=2000), + GitHubRepo("angular", "angular", languages=["typescript"], max_files=2000), + GitHubRepo("vercel", "next.js", languages=["javascript", "typescript"], max_files=2000), + GitHubRepo("microsoft", "TypeScript", languages=["typescript"], max_files=2000), + GitHubRepo("nodejs", "node", languages=["javascript", "cpp"], max_files=2000), + GitHubRepo("expressjs", "express", languages=["javascript"]), + GitHubRepo("lodash", "lodash", languages=["javascript"]), + GitHubRepo("axios", "axios", languages=["javascript"]), + GitHubRepo("chalk", "chalk", languages=["javascript"]), + + # Go + GitHubRepo("golang", "go", languages=["go"], max_files=2000), + GitHubRepo("gin-gonic", "gin", languages=["go"]), + GitHubRepo("labstack", "echo", languages=["go"]), + GitHubRepo("spf13", "cobra", languages=["go"]), + GitHubRepo("kubernetes", "kubernetes", languages=["go"], max_files=2000), + GitHubRepo("prometheus", "prometheus", languages=["go"], max_files=2000), + GitHubRepo("grafana", "grafana", languages=["go"], max_files=2000), + GitHubRepo("etcd-io", "etcd", languages=["go"], max_files=2000), + GitHubRepo("hashicorp", "terraform", languages=["go"], max_files=2000), + GitHubRepo("hashicorp", "vault", languages=["go"], max_files=2000), + GitHubRepo("docker", "compose", languages=["go"]), + GitHubRepo("cli", "cli", languages=["go"]), + + # Rust + GitHubRepo("rust-lang", "rust", languages=["rust"], max_files=2000), + GitHubRepo("rust-lang", "cargo", languages=["rust"], max_files=2000), + GitHubRepo("tokio-rs", "tokio", languages=["rust"], max_files=2000), + GitHubRepo("serde-rs", "serde", languages=["rust"]), + GitHubRepo("clap-rs", "clap", languages=["rust"]), + GitHubRepo("BurntSushi", "ripgrep", languages=["rust"]), + GitHubRepo("starship", "starship", languages=["rust"], max_files=2000), + + # C/C++ + GitHubRepo("redis", "redis", languages=["c"], max_files=2000), + GitHubRepo("sqlite", "sqlite", languages=["c"]), + GitHubRepo("curl", "curl", languages=["c"], max_files=2000), + GitHubRepo("nginx", "nginx", languages=["c"], max_files=2000), + GitHubRepo("openssl", "openssl", languages=["c"], max_files=2000), + + # Tools/CLI + GitHubRepo("junegunn", "fzf", languages=["go"]), + GitHubRepo("BurntSushi", "ripgrep", languages=["rust"]), + GitHubRepo("sharkdp", "bat", languages=["rust"]), + GitHubRepo("sharkdp", "fd", languages=["rust"]), + GitHubRepo("dalance", "procs", languages=["rust"]), + + # Documentation/Examples + GitHubRepo("realpython", "python-guide", languages=["python", "markdown"]), + GitHubRepo("ehmatthes", "pcc_2e", languages=["python"]), + GitHubRepo("thedaviddias", "Front-End-Checklist", languages=["markdown"]), + GitHubRepo("kamranahmedse", "developer-roadmap", languages=["markdown"]), +] diff --git a/nexus/data/collectors/huggingface_collector.py b/nexus/data/collectors/huggingface_collector.py new file mode 100644 index 0000000000000000000000000000000000000000..0f910a03c48d4e28a935e9bdc4fa920d9e7753c4 --- /dev/null +++ b/nexus/data/collectors/huggingface_collector.py @@ -0,0 +1,310 @@ +""" +HuggingFace Collector - Thu thập datasets từ HuggingFace Hub +============================================================= +Pull datasets từ HuggingFace Hub cho training Nexus Coder. + +Recommended datasets for code/text training: +- codeparrot/codeparrot-clean: Clean Python code +- GitHub CODE: Code from GitHub +- the-stack: Massive code dataset (3TB) +- oscar: Multilingual web text +- wikipedia: Wikipedia dumps +- openwebtext: Web text +- c4: Colossal Clean Crawled Corpus +- bookcorpus: Books +- arxiv: Scientific papers +- pubmed: Biomedical papers +""" +from __future__ import annotations + +import os +import json +import logging +from typing import List, Dict, Optional, Iterator, Any +from dataclasses import dataclass, field +from pathlib import Path + +logger = logging.getLogger(__name__) + + +@dataclass +class HFDataset: + """Thông tin một HuggingFace dataset.""" + name: str # e.g. "codeparrot/codeparrot-clean" + subset: Optional[str] = None + split: str = "train" + streaming: bool = True # Use streaming for large datasets + max_samples: int = 10000 + field_mapping: Dict[str, str] = field(default_factory=lambda: {"text": "text"}) + description: str = "" + language: Optional[str] = None # programming language for code datasets + size_gb: Optional[float] = None + + +# ============================================================================= +# Curated list of high-quality datasets for Nexus Coder training +# ============================================================================= + +CURATED_DATASETS: List[HFDataset] = [ + # === Code datasets === + HFDataset( + name="codeparrot/codeparrot-clean", + max_samples=50000, + language="python", + description="Clean Python code from GitHub (preprocessed)", + size_gb=15, + ), + HFDataset( + name="codeparrot/github-code", + max_samples=30000, + language="multiple", + description="Code from GitHub across multiple languages", + size_gb=115, + ), + HFDataset( + name="bigcode/the-stack-dedup", + max_samples=20000, + language="multiple", + description="Deduplicated code from The Stack v2 (3TB)", + size_gb=3000, + ), + HFDataset( + name="bigcode/the-stack-v2-train-full-ids", + max_samples=10000, + language="multiple", + description="The Stack v2 full training set", + size_gb=3000, + ), + HFDataset( + name="nampdn-ai/tiny-codes", + max_samples=30000, + language="multiple", + description="Small high-quality code samples with instructions", + size_gb=2, + ), + HFDataset( + name="HuggingFaceH4/CodeAlpaca_20K", + max_samples=20000, + language="python", + description="Code instruction dataset", + size_gb=0.1, + ), + + # === General text (Vietnamese + English) === + HFDataset( + name="wikimedia/wikipedia", + subset="20231101.vi", + max_samples=20000, + description="Vietnamese Wikipedia", + size_gb=2, + ), + HFDataset( + name="wikimedia/wikipedia", + subset="20231101.en", + max_samples=20000, + description="English Wikipedia", + size_gb=20, + ), + HFDataset( + name="allenai/c4", + subset="multilingual", + split="train", + max_samples=10000, + description="Colossal Clean Crawled Corpus (multilingual)", + size_gb=25000, + ), + HFDataset( + name="oscar-corpus/OSCAR-2301", + subset="vi", + max_samples=10000, + description="OSCAR Vietnamese web text", + size_gb=10, + ), + + # === Conversational / Instruction === + HFDataset( + name="HuggingFaceH4/ultrachat_200k", + max_samples=20000, + description="High-quality multi-turn chat data", + size_gb=8, + ), + HFDataset( + name="Open-Orca/OpenOrca", + max_samples=15000, + description="GPT-4 augmented FLAN instructions", + size_gb=50, + ), + HFDataset( + name="teknium/OpenHermes-2.5", + max_samples=20000, + description="1M instruction samples", + size_gb=5, + ), + HFDataset( + name="databricks/databricks-dolly-15k", + max_samples=15000, + description="Human-generated instruction data", + size_gb=0.2, + ), + HFDataset( + name="allenai/RLVR-Chat", + max_samples=10000, + description="Reinforcement Learning from Verifiable Rewards chat data", + size_gb=2, + ), + + # === Math/Reasoning === + HFDataset( + name="meta-math/MetaMathQA", + max_samples=20000, + description="Math Q&A with step-by-step solutions", + size_gb=1, + ), + HFDataset( + name="gsm8k", + max_samples=8000, + description="Grade School Math 8K", + size_gb=0.01, + ), + HFDataset( + name="lighteval/MATH", + max_samples=10000, + description="Competition math problems", + size_gb=0.05, + ), + + # === Scientific === + HFDataset( + name="allenai/sciq", + max_samples=13000, + description="Science exam questions", + size_gb=0.05, + ), + HFDataset( + name="allenai/openbookqa", + max_samples=5000, + description="Open-book science Q&A", + size_gb=0.02, + ), + + # === Vietnamese specific === + HFDataset( + name="vietgpt/news_corpus", + max_samples=10000, + description="Vietnamese news corpus", + size_gb=2, + ), + HFDataset( + name="PhoATC", + max_samples=5000, + description="Vietnamese text classification", + size_gb=0.1, + ), +] + + +class HuggingFaceCollector: + """Collect training data từ HuggingFace Hub. + + Usage: + collector = HuggingFaceCollector(cache_dir="./data_cache/hf") + for sample in collector.collect(CURATED_DATASETS[:3]): + print(sample["text"][:100]) + """ + + def __init__( + self, + cache_dir: str = "./data_cache/hf", + token: Optional[str] = None, + ): + self.cache_dir = cache_dir + self.token = token or os.environ.get("HF_TOKEN") or os.environ.get("HUGGING_FACE_HUB_TOKEN") + os.makedirs(cache_dir, exist_ok=True) + + def collect(self, datasets: List[HFDataset]) -> Iterator[Dict[str, Any]]: + """Collect samples từ list of HF datasets. + + Yields: + Dict with keys: text, source, language, metadata + """ + try: + from datasets import load_dataset + except ImportError: + logger.error("datasets lib not installed. Run: pip install datasets") + return + + for ds in datasets: + try: + yield from self._collect_dataset(ds, load_dataset) + except Exception as e: + logger.error(f"Failed to collect {ds.name}: {e}") + continue + + def _collect_dataset( + self, + ds: HFDataset, + load_fn, + ) -> Iterator[Dict[str, Any]]: + """Collect từ một dataset.""" + logger.info(f"Loading {ds.name} ({ds.subset or 'default'})...") + + try: + if ds.streaming: + dataset = load_fn( + ds.name, + name=ds.subset, + split=ds.split, + streaming=True, + token=self.token, + ) + else: + dataset = load_fn( + ds.name, + name=ds.subset, + split=ds.split, + token=self.token, + cache_dir=self.cache_dir, + ) + except Exception as e: + logger.error(f"Failed to load {ds.name}: {e}") + return + + count = 0 + text_field = ds.field_mapping.get("text", "text") + + for item in dataset: + if count >= ds.max_samples: + break + + # Extract text using field mapping + text = item.get(text_field) or item.get("text") or item.get("content") or "" + + if not text or not isinstance(text, str): + # Try concatenating fields + text = " ".join(str(v) for v in item.values() if isinstance(v, str)) + + if not text or len(text) < 50: + continue + + yield { + "text": text, + "source": ds.name, + "language": ds.language or "text", + "metadata": { + "dataset": ds.name, + "subset": ds.subset, + "split": ds.split, + "original_size": len(text), + }, + } + count += 1 + + logger.info(f"Collected {count} samples from {ds.name}") + + def list_available(self) -> List[HFDataset]: + """Return curated list of datasets.""" + return CURATED_DATASETS + + def estimate_total_size(self, datasets: List[HFDataset]) -> float: + """Estimate total size in GB.""" + return sum(ds.size_gb or 0 for ds in datasets) diff --git a/nexus/data/collectors/python_alpaca_collector.py b/nexus/data/collectors/python_alpaca_collector.py new file mode 100644 index 0000000000000000000000000000000000000000..2791902ec52bb0ff7090c3f0b7a603b6eacbad89 --- /dev/null +++ b/nexus/data/collectors/python_alpaca_collector.py @@ -0,0 +1,117 @@ +""" +Python-Alpaca Collector for Nexus Coder v0.3 +============================================= +Aggregates multiple high-quality Python instruction-tuning datasets. + +Sources (all on HuggingFace): + - sahil2801/codealpaca ~20K samples + - HuggingFaceH4/CodeAlpaca_20K ~20K + - nickroany/Evol-Instruct-Code ~15K + - TheBloke/CodeAlpaca-13B ~5K + - codeparrot/codeparrot-clean ~50K (filterable) + - nampdn-ai/tiny-codes ~50K (filterable) + +Output: unified JSONL with Nexus format {system, user, assistant}. +Converts Alpaca-style {instruction, input, output} → unified via +nexus.integrations.llamafactory.alpaca_to_nexus. + +Author: Hieu Louis (2026) +""" +from __future__ import annotations + +import json +import os +from typing import Dict, Iterator, List, Optional + +# We import the converter for type hints only — actual import at runtime +# to keep the module importable when llamafactory deps are missing. +try: + from ...integrations.llamafactory import convert_to_nexus + _HAS_CONVERTER = True +except Exception: + _HAS_CONVERTER = False + + +DEFAULT_SOURCES = [ + {"name": "sahil2801/codealpaca", "max_samples": 20000}, + {"name": "HuggingFaceH4/CodeAlpaca_20K", "max_samples": 20000}, + {"name": "nickroany/Evol-Instruct-Code", "max_samples": 15000}, + {"name": "TheBloke/CodeAlpaca-13B", "max_samples": 5000}, + {"name": "codeparrot/codeparrot-clean", "max_samples": 50000, "is_completion": True}, + {"name": "nampdn-ai/tiny-codes", "max_samples": 50000}, +] + + +class PythonAlpacaCollector: + """Aggregate Python instruction datasets.""" + + def __init__( + self, + cache_dir: str = "./data_cache/python_alpaca", + sources: Optional[List[Dict]] = None, + ): + self.cache_dir = cache_dir + self.sources = sources or DEFAULT_SOURCES + os.makedirs(cache_dir, exist_ok=True) + + def _iter_source(self, source: Dict) -> Iterator[Dict]: + name = source["name"] + max_samples = source.get("max_samples", 10000) + is_completion = source.get("is_completion", False) + try: + from datasets import load_dataset + except ImportError: + return + try: + ds = load_dataset(name, split="train", streaming=True) + except Exception: + return + count = 0 + for example in ds: + if count >= max_samples: + break + # Normalize to Nexus format + try: + if _HAS_CONVERTER: + turns = convert_to_nexus(example) + else: + # Inline fallback for Alpaca format + turns = [{ + "system": example.get("system_prompt", ""), + "user": example.get("instruction", ""), + "assistant": example.get("output", ""), + }] + for turn in turns: + if not turn.get("user") or not turn.get("assistant"): + continue + yield { + "source": name, + "system": turn.get("system", ""), + "user": turn["user"], + "assistant": turn["assistant"], + } + count += 1 + if count >= max_samples: + break + except Exception: + continue + + def __iter__(self) -> Iterator[Dict]: + for source in self.sources: + yield from self._iter_source(source) + + def collect(self, output_dir: Optional[str] = None) -> str: + """Collect and write JSONL. Returns output path.""" + output_dir = output_dir or self.cache_dir + os.makedirs(output_dir, exist_ok=True) + output_path = os.path.join(output_dir, "python_alpaca.jsonl") + total = 0 + with open(output_path, "w", encoding="utf-8") as f: + for sample in self: + f.write(json.dumps(sample, ensure_ascii=False) + "\n") + total += 1 + print(f"[PythonAlpacaCollector] Collected {total} samples → {output_path}") + return output_path + + +__all__ = ["PythonAlpacaCollector", "DEFAULT_SOURCES"] diff --git a/nexus/data/collectors/stackoverflow_collector.py b/nexus/data/collectors/stackoverflow_collector.py new file mode 100644 index 0000000000000000000000000000000000000000..151df0cefb83e2156484491bf35c6146164a56f1 --- /dev/null +++ b/nexus/data/collectors/stackoverflow_collector.py @@ -0,0 +1,250 @@ +""" +StackOverflow Collector - Thu thập Q&A từ StackOverflow +========================================================= +""" +from __future__ import annotations + +import logging +import urllib.request +import urllib.parse +import json +import time +from typing import List, Dict, Optional, Iterator, Any +from dataclasses import dataclass + +logger = logging.getLogger(__name__) + + +@dataclass +class SOQuestion: + """Một StackOverflow question.""" + question_id: int + title: str + body: str + tags: List[str] + score: int + answer_count: int + accepted_answer_id: Optional[int] = None + answers: List[Dict] = None + + +# v0.4 fix: expose at module level (was inside the class, broke `from ... import CURATED_TAGS`) +CURATED_TAGS = [ + "python", "javascript", "java", "c#", "php", "android", + "html", "jquery", "c++", "css", "ios", "mysql", + "sql", "node.js", "reactjs", "ruby-on-rails", "vue.js", + "typescript", "docker", "git", "go", "rust", + "machine-learning", "deep-learning", "pytorch", "tensorflow", + "pandas", "numpy", "regex", "algorithm", "data-structures", + "unit-testing", "debugging", "performance", "security", +] + + +class StackOverflowCollector: + """Collect Q&A từ StackOverflow API. + + StackOverflow API: 10000 requests/day without key, 50000 with key. + Rate limit: 30 requests/second. + """ + + BASE_URL = "https://api.stackexchange.com/2.3" + + # Backward-compat alias (deprecation: prefer module-level CURATED_TAGS) + CURATED_TAGS = CURATED_TAGS + + def __init__( + self, + key: Optional[str] = None, + access_token: Optional[str] = None, + page_size: int = 100, + ): + self.key = key + self.access_token = access_token + self.page_size = min(page_size, 100) + + def search( + self, + tag: str, + max_results: int = 500, + min_score: int = 5, + sort: str = "votes", + ) -> List[SOQuestion]: + """Search questions by tag. + + Args: + tag: Tag to filter (e.g. "python") + max_results: Max questions to return + min_score: Minimum question score + sort: "votes", "creation", "activity" + """ + questions = [] + page = 1 + + while len(questions) < max_results and page <= 50: # API limit: 50 pages + params = { + "order": "desc", + "sort": sort, + "tagged": tag, + "site": "stackoverflow", + "pagesize": str(self.page_size), + "page": str(page), + "filter": "withbody", # Include body + "min": str(min_score), + } + if self.key: + params["key"] = self.key + if self.access_token: + params["access_token"] = self.access_token + + url = f"{self.BASE_URL}/questions?{urllib.parse.urlencode(params)}" + + try: + req = urllib.request.Request(url, headers={ + "Accept-Encoding": "gzip", + "User-Agent": "NexusCoder-Collector/0.2", + }) + with urllib.request.urlopen(req, timeout=30) as response: + # Handle gzip + if response.headers.get("Content-Encoding") == "gzip": + import gzip + data = json.loads(gzip.decompress(response.read()).decode()) + else: + data = json.loads(response.read().decode()) + + items = data.get("items", []) + if not items: + break + + for item in items: + questions.append(SOQuestion( + question_id=item["question_id"], + title=item["title"], + body=item.get("body", ""), + tags=item.get("tags", []), + score=item.get("score", 0), + answer_count=item.get("answer_count", 0), + accepted_answer_id=item.get("accepted_answer_id"), + )) + + # Check if more pages + if not data.get("has_more", False): + break + + # Backoff if needed + if data.get("backoff"): + time.sleep(data["backoff"]) + + page += 1 + time.sleep(0.5) # Polite delay + + except Exception as e: + logger.error(f"SO search failed: {e}") + break + + return questions[:max_results] + + def get_answers(self, question_ids: List[int]) -> Dict[int, List[Dict]]: + """Get answers for multiple questions.""" + if not question_ids: + return {} + + ids_str = ";".join(str(qid) for qid in question_ids[:100]) # Max 100 ids + params = { + "order": "desc", + "sort": "votes", + "site": "stackoverflow", + "filter": "withbody", + } + if self.key: + params["key"] = self.key + + url = f"{self.BASE_URL}/questions/{ids_str}/answers?{urllib.parse.urlencode(params)}" + + try: + req = urllib.request.Request(url, headers={ + "Accept-Encoding": "gzip", + "User-Agent": "NexusCoder-Collector/0.2", + }) + with urllib.request.urlopen(req, timeout=30) as response: + if response.headers.get("Content-Encoding") == "gzip": + import gzip + data = json.loads(gzip.decompress(response.read()).decode()) + else: + data = json.loads(response.read().decode()) + + answers_by_q = {} + for ans in data.get("items", []): + qid = ans["question_id"] + if qid not in answers_by_q: + answers_by_q[qid] = [] + answers_by_q[qid].append({ + "answer_id": ans["answer_id"], + "body": ans.get("body", ""), + "score": ans.get("score", 0), + "is_accepted": ans.get("is_accepted", False), + }) + + return answers_by_q + except Exception as e: + logger.error(f"SO get_answers failed: {e}") + return {} + + def collect( + self, + tags: Optional[List[str]] = None, + max_per_tag: int = 100, + include_answers: bool = True, + ) -> Iterator[Dict[str, Any]]: + """Collect Q&A pairs as training samples. + + Yields: + Dict with keys: text (formatted Q&A), source, language, metadata + """ + tags = tags or self.CURATED_TAGS[:10] + + for tag in tags: + logger.info(f"Collecting SO tag: {tag}") + questions = self.search(tag, max_results=max_per_tag) + + if include_answers and questions: + qids = [q.question_id for q in questions if q.accepted_answer_id] + answers_by_q = self.get_answers(qids) + else: + answers_by_q = {} + + for q in questions: + # Format as Q&A pair + answer_text = "" + if q.question_id in answers_by_q: + accepted = [a for a in answers_by_q[q.question_id] if a["is_accepted"]] + if accepted: + answer_text = accepted[0]["body"] + elif answers_by_q[q.question_id]: + answer_text = answers_by_q[q.question_id][0]["body"] + + if not answer_text: + continue + + # Strip HTML tags (simple) + import re + q_body_clean = re.sub(r"<[^>]+>", "", q.body) + a_clean = re.sub(r"<[^>]+>", "", answer_text) + + text = ( + f"Question: {q.title}\n\n" + f"Tags: {', '.join(q.tags)}\n\n" + f"{q_body_clean}\n\n" + f"Answer:\n{a_clean}" + ) + + yield { + "text": text, + "source": "stackoverflow", + "language": "en", + "metadata": { + "question_id": q.question_id, + "tags": q.tags, + "score": q.score, + "title": q.title, + }, + } diff --git a/nexus/data/collectors/starcoder2_collector.py b/nexus/data/collectors/starcoder2_collector.py new file mode 100644 index 0000000000000000000000000000000000000000..5c9bc5afa47a8ded1cbbbf4eafdf141a5a55739b --- /dev/null +++ b/nexus/data/collectors/starcoder2_collector.py @@ -0,0 +1,186 @@ +""" +StarCoder2-data Collector for Nexus Coder v0.3 +=============================================== +Pulls from BigCode's StarCoder2 training data (github-code, commits, jupyter). + +Components: + - github_code: raw code files from GitHub (subset of The-Stack v2) + - github_commits: commit diffs — great for code-editing / instruction tasks + - github_jupyter: notebook cells with markdown + code interleaved + +Each component has different schema; this collector unifies them into the +Nexus format: {source, lang, content, metadata}. + +Reference: + BigCode. "StarCoder 2 and The Stack v2: Building the Next Generation of + Transparent Code Models." + https://huggingface.co/datasets/bigcode/starcoder2data + +Author: Hieu Louis (2026) +""" +from __future__ import annotations + +import json +import os +from typing import Dict, Iterator, List, Optional + + +COMPONENT_DATASETS = { + "github_code": "bigcode/starcoder2data", + "github_commits": "bigcode/starcoder2data", + "github_jupyter": "bigcode/starcoder2data", +} + +SUPPORTED_LANGS = [ + "python", "javascript", "typescript", "java", + "go", "rust", "c", "cpp", +] + + +class StarCoder2Collector: + """Collect from StarCoder2 training data.""" + + def __init__( + self, + cache_dir: str = "./data_cache/starcoder2", + components: Optional[List[str]] = None, + max_samples_per_component: int = 20000, + languages: Optional[List[str]] = None, + streaming: bool = True, + ): + self.cache_dir = cache_dir + self.components = components or list(COMPONENT_DATASETS.keys()) + self.max_samples_per_component = max_samples_per_component + self.languages = languages or SUPPORTED_LANGS + self.streaming = streaming + os.makedirs(cache_dir, exist_ok=True) + + def _iter_github_code(self) -> Iterator[Dict]: + """Iterate github_code component.""" + try: + from datasets import load_dataset + except ImportError: + return + for lang in self.languages: + count = 0 + try: + ds = load_dataset( + "bigcode/starcoder2data", + split="train", + streaming=self.streaming, + data_dir=f"data/{lang}", + ) + except Exception: + continue + for example in ds: + if count >= self.max_samples_per_component // len(self.languages): + break + content = example.get("content", "") + if not content or len(content) < 50: + continue + yield { + "source": "starcoder2_github_code", + "lang": lang, + "content": content, + "metadata": { + "repo": example.get("repository", ""), + "path": example.get("path", ""), + "size": example.get("size", 0), + "license": example.get("license", ""), + }, + } + count += 1 + + def _iter_github_commits(self) -> Iterator[Dict]: + """Iterate github_commits component (commit diffs).""" + try: + from datasets import load_dataset + except ImportError: + return + count = 0 + try: + ds = load_dataset( + "bigcode/starcoder2data", + split="train", + streaming=self.streaming, + name="commits", + ) + except Exception: + return + for example in ds: + if count >= self.max_samples_per_component: + break + diff = example.get("diff", "") or example.get("content", "") + if not diff or len(diff) < 50: + continue + yield { + "source": "starcoder2_commits", + "lang": example.get("language", "unknown"), + "content": diff, + "metadata": { + "commit": example.get("commit", ""), + "repo": example.get("repository", ""), + "author": example.get("author", ""), + }, + } + count += 1 + + def _iter_github_jupyter(self) -> Iterator[Dict]: + """Iterate github_jupyter component (notebook cells).""" + try: + from datasets import load_dataset + except ImportError: + return + count = 0 + try: + ds = load_dataset( + "bigcode/starcoder2data", + split="train", + streaming=self.streaming, + name="jupyter", + ) + except Exception: + return + for example in ds: + if count >= self.max_samples_per_component: + break + content = example.get("content", "") + if not content or len(content) < 50: + continue + yield { + "source": "starcoder2_jupyter", + "lang": "python", + "content": content, + "metadata": { + "repo": example.get("repository", ""), + "notebook_path": example.get("path", ""), + "cell_type": example.get("cell_type", ""), + }, + } + count += 1 + + def __iter__(self) -> Iterator[Dict]: + """Stream samples from all enabled components.""" + for component in self.components: + if component == "github_code": + yield from self._iter_github_code() + elif component == "github_commits": + yield from self._iter_github_commits() + elif component == "github_jupyter": + yield from self._iter_github_jupyter() + + def collect(self, output_dir: Optional[str] = None) -> str: + """Collect all samples and write to JSONL. Returns the output file path.""" + output_dir = output_dir or self.cache_dir + os.makedirs(output_dir, exist_ok=True) + output_path = os.path.join(output_dir, "starcoder2.jsonl") + total = 0 + with open(output_path, "w", encoding="utf-8") as f: + for sample in self: + f.write(json.dumps(sample, ensure_ascii=False) + "\n") + total += 1 + print(f"[StarCoder2Collector] Collected {total} samples → {output_path}") + return output_path + + +__all__ = ["StarCoder2Collector", "COMPONENT_DATASETS", "SUPPORTED_LANGS"] diff --git a/nexus/data/collectors/the_stack_collector.py b/nexus/data/collectors/the_stack_collector.py new file mode 100644 index 0000000000000000000000000000000000000000..5f90bc653d44cf06b726ee433ec7441dee11b276 --- /dev/null +++ b/nexus/data/collectors/the_stack_collector.py @@ -0,0 +1,131 @@ +""" +The-Stack v2 Collector for Nexus Coder v0.3 +============================================ +Pulls code samples from BigCode's The-Stack v2 dataset on HuggingFace. + +The-Stack v2 is a massive deduplicated code corpus covering ~600 programming +languages, collected from GitHub repos with permissive licenses. + +This collector: + - Streams samples lazily via `datasets` library (lazy import) + - Filters by language (Python, JS, TS, Go, Rust, etc.) + - Applies license filter (only MIT/Apache/BSD/MPL) + - Writes to JSONL with metadata {lang, license, repo, path, content} + +Reference: + BigCode. "The Stack v2: A Comprehensive Multilingual Code Corpus." + https://huggingface.co/datasets/bigcode/the-stack-v2-train-full-ids + +Author: Hieu Louis (2026) +""" +from __future__ import annotations + +import json +import os +from typing import Dict, Iterator, List, Optional + + +# Curated language list (subset of v2's ~600 languages) +SUPPORTED_LANGUAGES = [ + "python", "javascript", "typescript", "java", "go", "rust", + "c", "cpp", "csharp", "ruby", "php", "swift", "kotlin", + "scala", "shell", "sql", "html", "css", +] + +# Permissive licenses (allowlist) +PERMISSIVE_LICENSES = { + "mit", "apache-2.0", "bsd-3-clause", "bsd-2-clause", + "mpl-2.0", "unlicense", "isc", "0bsd", +} + + +class TheStackCollector: + """Collect code samples from The-Stack v2.""" + + DATASET_NAME = "bigcode/the-stack-v2-train-full-ids" + + def __init__( + self, + cache_dir: str = "./data_cache/the_stack", + languages: Optional[List[str]] = None, + max_samples_per_language: int = 5000, + min_stars: int = 0, + license_filter: Optional[List[str]] = None, + streaming: bool = True, + ): + self.cache_dir = cache_dir + self.languages = languages or SUPPORTED_LANGUAGES + self.max_samples_per_language = max_samples_per_language + self.min_stars = min_stars + self.license_filter = set(license_filter) if license_filter else PERMISSIVE_LICENSES + self.streaming = streaming + os.makedirs(cache_dir, exist_ok=True) + + def __iter__(self) -> Iterator[Dict]: + """Stream samples lazily from The-Stack v2. + Yields dicts: {lang, license, repo, path, size, content}. + """ + try: + from datasets import load_dataset # lazy import + except ImportError as e: + raise ImportError( + "The `datasets` package is required. Install with: pip install datasets" + ) from e + + for lang in self.languages: + count = 0 + try: + ds = load_dataset( + self.DATASET_NAME, + split="train", + streaming=self.streaming, + data_dir=f"data/{lang}", + ) + except Exception: + continue + for example in ds: + if count >= self.max_samples_per_language: + break + # Apply filters + stars = example.get("stars", 0) or 0 + if stars < self.min_stars: + continue + license_ = (example.get("license") or "").lower() + if license_ and license_ not in self.license_filter: + continue + content = example.get("content", "") + if not content or len(content) < 50: + continue + yield { + "lang": lang, + "license": license_, + "repo": example.get("repository", ""), + "path": example.get("path", ""), + "size": example.get("size", len(content)), + "stars": stars, + "content": content, + } + count += 1 + + def collect(self, output_dir: Optional[str] = None) -> str: + """Collect all samples and write to JSONL. Returns the output file path.""" + output_dir = output_dir or self.cache_dir + os.makedirs(output_dir, exist_ok=True) + output_path = os.path.join(output_dir, "the_stack_v2.jsonl") + total = 0 + with open(output_path, "w", encoding="utf-8") as f: + for sample in self: + f.write(json.dumps(sample, ensure_ascii=False) + "\n") + total += 1 + print(f"[TheStackCollector] Collected {total} samples → {output_path}") + return output_path + + def stats(self) -> Dict[str, int]: + """Return per-language sample counts (calls collect if not yet run).""" + counts: Dict[str, int] = {lang: 0 for lang in self.languages} + for sample in self: + counts[sample["lang"]] = counts.get(sample["lang"], 0) + 1 + return counts + + +__all__ = ["TheStackCollector", "SUPPORTED_LANGUAGES", "PERMISSIVE_LICENSES"] diff --git a/nexus/data/collectors/wikipedia_collector.py b/nexus/data/collectors/wikipedia_collector.py new file mode 100644 index 0000000000000000000000000000000000000000..5b5c450259ed456def8c98c50d3e8c6d29212500 --- /dev/null +++ b/nexus/data/collectors/wikipedia_collector.py @@ -0,0 +1,164 @@ +""" +Wikipedia Collector - Thu thập dữ liệu từ Wikipedia +==================================================== +""" +from __future__ import annotations + +import logging +import urllib.request +import urllib.parse +import json +from typing import List, Dict, Optional, Iterator, Any +from dataclasses import dataclass + +logger = logging.getLogger(__name__) + + +@dataclass +class WikiArticle: + """Một Wikipedia article.""" + title: str + content: str + url: str + language: str + categories: List[str] + + +class WikipediaCollector: + """Collect articles từ Wikipedia API. + + Supports Vietnamese (vi) and English (en) Wikipedia. + """ + + BASE_URLS = { + "vi": "https://vi.wikipedia.org/w/api.php", + "en": "https://en.wikipedia.org/w/api.php", + } + + RANDOM_TOPICS = { + "vi": [ + "Trí tuệ nhân tạo", "Học máy", "Mạng nơ-ron nhân tạo", + "Python (ngôn ngữ lập trình)", "JavaScript", "Linux", + "Cơ sở dữ liệu", "Thuật toán", "Cấu trúc dữ liệu", + "Lập trình hướng đối tượng", "API", "JSON", "Git", + "Hệ điều hành", "Máy học sâu", "Xử lý ngôn ngữ tự nhiên", + "Học sâu", "Big data", "Điện toán đám mây", + ], + "en": [ + "Artificial intelligence", "Machine learning", "Neural network", + "Python (programming language)", "JavaScript", "Linux", + "Database", "Algorithm", "Data structure", + "Object-oriented programming", "API", "JSON", "Git", + "Operating system", "Deep learning", "Natural language processing", + "Big data", "Cloud computing", "Transformer (deep learning model)", + "Large language model", "GPT-4", "BERT (language model)", + ], + } + + def __init__(self, language: str = "vi"): + self.language = language + self.base_url = self.BASE_URLS.get(language, self.BASE_URLS["en"]) + + def get_article(self, title: str) -> Optional[WikiArticle]: + """Lấy nội dung một Wikipedia article theo title.""" + params = { + "action": "query", + "titles": title, + "prop": "extracts|categories", + "exintro": "false", + "explaintext": "true", + "cllimit": "10", + "format": "json", + "redirects": "1", + } + + url = f"{self.base_url}?{urllib.parse.urlencode(params)}" + + try: + req = urllib.request.Request(url, headers={"User-Agent": "NexusCoder-Collector/0.2"}) + with urllib.request.urlopen(req, timeout=30) as response: + data = json.loads(response.read().decode()) + + pages = data.get("query", {}).get("pages", {}) + if not pages: + return None + + page = list(pages.values())[0] + if "missing" in page: + return None + + content = page.get("extract", "") + if not content or len(content) < 100: + return None + + categories = [] + for cat in page.get("categories", []): + categories.append(cat["title"].replace("Category:", "")) + + title_resolved = page.get("title", title) + url_resolved = f"https://{self.language}.wikipedia.org/wiki/{urllib.parse.quote(title_resolved.replace(' ', '_'))}" + + return WikiArticle( + title=title_resolved, + content=content, + url=url_resolved, + language=self.language, + categories=categories, + ) + except Exception as e: + logger.error(f"Wikipedia fetch failed for '{title}': {e}") + return None + + def collect( + self, + topics: Optional[List[str]] = None, + max_per_topic: int = 1, + ) -> Iterator[Dict[str, Any]]: + """Collect articles, yield as text samples.""" + topics = topics or self.RANDOM_TOPICS.get(self.language, self.RANDOM_TOPICS["en"]) + + for topic in topics: + article = self.get_article(topic) + if article: + yield { + "text": f"# {article.title}\n\n{article.content}", + "source": f"wikipedia_{self.language}", + "language": self.language, + "metadata": { + "title": article.title, + "url": article.url, + "categories": article.categories, + }, + } + + def collect_random(self, count: int = 100) -> Iterator[Dict[str, Any]]: + """Collect random articles via Wikipedia API.""" + params = { + "action": "query", + "list": "random", + "rnnamespace": "0", # Main namespace + "rnlimit": str(count), + "format": "json", + } + + url = f"{self.base_url}?{urllib.parse.urlencode(params)}" + + try: + req = urllib.request.Request(url, headers={"User-Agent": "NexusCoder-Collector/0.2"}) + with urllib.request.urlopen(req, timeout=30) as response: + data = json.loads(response.read().decode()) + + for item in data.get("query", {}).get("random", []): + article = self.get_article(item["title"]) + if article: + yield { + "text": f"# {article.title}\n\n{article.content}", + "source": f"wikipedia_{self.language}_random", + "language": self.language, + "metadata": { + "title": article.title, + "url": article.url, + }, + } + except Exception as e: + logger.error(f"Wikipedia random failed: {e}") diff --git a/nexus/data/curriculum.py b/nexus/data/curriculum.py new file mode 100644 index 0000000000000000000000000000000000000000..685c729f0b85278cd1c4742d848ed2c3e9d459ab --- /dev/null +++ b/nexus/data/curriculum.py @@ -0,0 +1,162 @@ +"""Curriculum Learning - Học theo lộ trình từ dễ đến khó.""" +from __future__ import annotations + +from typing import List, Dict, Any, Iterator, Optional, Callable +from dataclasses import dataclass, field +from enum import Enum + + +class Difficulty(str, Enum): + """Mức độ khó của samples.""" + EASY = "easy" # Short text, simple vocabulary + MEDIUM = "medium" # Standard length, normal vocabulary + HARD = "hard" # Long text, technical, complex + EXPERT = "expert" # Very long, very technical, multi-step + + +@dataclass +class CurriculumStage: + """Một stage trong curriculum learning.""" + name: str + difficulty: Difficulty + min_length: int = 0 + max_length: int = 10000 + min_quality: float = 0.5 + weight: float = 1.0 # Sampling weight + description: str = "" + source_filter: Optional[List[str]] = None # Only from these sources + + +class CurriculumLearning: + """Curriculum learning scheduler. + + Stage 1 (EASY): Short samples, basic vocabulary + Stage 2 (MEDIUM): Standard samples + Stage 3 (HARD): Long technical samples + Stage 4 (EXPERT): Very long, multi-step reasoning + + Usage: + curr = CurriculumLearning() + for stage in curr.stages: + samples = curr.get_samples_for_stage(stage, all_samples) + train_one_epoch(model, samples) + """ + + DEFAULT_STAGES = [ + CurriculumStage( + name="stage_1_basics", + difficulty=Difficulty.EASY, + min_length=50, + max_length=500, + min_quality=0.7, + weight=1.0, + description="Short basic text - vocabulary building", + ), + CurriculumStage( + name="stage_2_standard", + difficulty=Difficulty.MEDIUM, + min_length=500, + max_length=5000, + min_quality=0.6, + weight=1.0, + description="Standard length text - grammar and reasoning", + ), + CurriculumStage( + name="stage_3_technical", + difficulty=Difficulty.HARD, + min_length=5000, + max_length=30000, + min_quality=0.7, + weight=0.8, + description="Long technical content - deep understanding", + ), + CurriculumStage( + name="stage_4_expert", + difficulty=Difficulty.EXPERT, + min_length=30000, + max_length=100000, + min_quality=0.8, + weight=0.5, + description="Expert-level multi-step reasoning", + ), + ] + + def __init__(self, stages: Optional[List[CurriculumStage]] = None): + self.stages = stages or self.DEFAULT_STAGES + + def classify_sample(self, sample: Dict[str, Any]) -> Difficulty: + """Classify sample into difficulty level.""" + text = sample.get("text", "") + length = len(text) + quality = sample.get("metadata", {}).get("quality", {}).get("score", 0.5) + + if length < 500 and quality >= 0.7: + return Difficulty.EASY + elif length < 5000 and quality >= 0.6: + return Difficulty.MEDIUM + elif length < 30000 and quality >= 0.7: + return Difficulty.HARD + else: + return Difficulty.EXPERT + + def get_samples_for_stage( + self, + stage: CurriculumStage, + samples: List[Dict[str, Any]], + ) -> List[Dict[str, Any]]: + """Filter samples for a specific stage.""" + result = [] + for sample in samples: + text = sample.get("text", "") + length = len(text) + quality = sample.get("metadata", {}).get("quality", {}).get("score", 0.5) + + # Length filter + if not (stage.min_length <= length <= stage.max_length): + continue + + # Quality filter + if quality < stage.min_quality: + continue + + # Source filter + if stage.source_filter: + source = sample.get("source", "") + if source not in stage.source_filter: + continue + + result.append(sample) + + return result + + def get_curriculum_schedule( + self, + total_steps: int, + num_stages: Optional[int] = None, + ) -> List[Dict[str, Any]]: + """Generate training schedule. + + Returns list of {stage, start_step, end_step, samples_ratio}. + """ + num_stages = num_stages or len(self.stages) + stages = self.stages[:num_stages] + + # Allocate steps to stages (more steps to harder stages) + total_weight = sum(s.weight for s in stages) + schedule = [] + current_step = 0 + + for stage in stages: + stage_steps = int(total_steps * stage.weight / total_weight) + schedule.append({ + "stage": stage.name, + "difficulty": stage.difficulty.value, + "start_step": current_step, + "end_step": current_step + stage_steps, + "steps": stage_steps, + "weight": stage.weight, + "description": stage.description, + }) + current_step += stage_steps + + return schedule diff --git a/nexus/data/processors/__init__.py b/nexus/data/processors/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..8ca70dd04052bdec9ad203842bf6351387b95a2d --- /dev/null +++ b/nexus/data/processors/__init__.py @@ -0,0 +1,31 @@ +"""Data processors package (v0.3 expanded). + +v0.2: TextCleaner, Deduplicator, QualityFilter, CodeFormatter +v0.3: + LanguageIdProcessor, CodeQualityProcessor +""" +from .cleaner import TextCleaner +from .deduplicator import Deduplicator +from .quality_filter import QualityFilter +from .code_formatter import CodeFormatter + +# v0.3 NEW +try: + from .language_id import LanguageIdProcessor +except ImportError: + LanguageIdProcessor = None # type: ignore + +try: + from .code_quality import CodeQualityProcessor +except ImportError: + CodeQualityProcessor = None # type: ignore + + +__all__ = [ + "TextCleaner", + "Deduplicator", + "QualityFilter", + "CodeFormatter", + # v0.3 NEW + "LanguageIdProcessor", + "CodeQualityProcessor", +] diff --git a/nexus/data/processors/cleaner.py b/nexus/data/processors/cleaner.py new file mode 100644 index 0000000000000000000000000000000000000000..5247d92143e969c7cd0d12cd581d2e27d6832481 --- /dev/null +++ b/nexus/data/processors/cleaner.py @@ -0,0 +1,131 @@ +"""Text Cleaner - Làm sạch text data.""" +from __future__ import annotations + +import re +import html +from typing import Dict, Any, List +from dataclasses import dataclass + + +@dataclass +class CleanerConfig: + """Config cho TextCleaner.""" + remove_html: bool = True + remove_urls: bool = False + remove_emojis: bool = False + normalize_whitespace: bool = True + normalize_unicode: bool = True + remove_control_chars: bool = True + min_length: int = 50 + max_length: int = 100000 + fix_encoding: bool = True + + +class TextCleaner: + """Làm sạch text data cho training. + + Usage: + cleaner = TextCleaner() + cleaned = cleaner.clean("some messy text...") + """ + + # Common patterns + HTML_TAG_RE = re.compile(r"<[^>]+>") + URL_RE = re.compile(r"https?://\S+|www\.\S+") + MULTI_SPACE_RE = re.compile(r"[ \t]+") + MULTI_NEWLINE_RE = re.compile(r"\n{3,}") + CONTROL_CHARS_RE = re.compile(r"[\x00-\x08\x0b\x0c\x0e-\x1f\x7f-\x9f]") + EMOJI_RE = re.compile( + "[" + "\U0001F600-\U0001F64F" + "\U0001F300-\U0001F5FF" + "\U0001F680-\U0001F6FF" + "\U0001F1E0-\U0001F1FF" + "\U00002702-\U000027B0" + "\U000024C2-\U0001F251" + "]+", + flags=re.UNICODE, + ) + + def __init__(self, config: CleanerConfig = None): + self.config = config or CleanerConfig() + + def clean(self, text: str) -> str: + """Clean a single text.""" + if not text or not isinstance(text, str): + return "" + + cfg = self.config + + # Fix encoding issues + if cfg.fix_encoding: + text = text.replace("\ufeff", "").replace("\u200b", "") + + # Normalize unicode + if cfg.normalize_unicode: + import unicodedata + text = unicodedata.normalize("NFC", text) + + # Remove control characters + if cfg.remove_control_chars: + text = self.CONTROL_CHARS_RE.sub("", text) + + # Decode HTML entities + text = html.unescape(text) + + # Remove HTML tags + if cfg.remove_html: + text = self.HTML_TAG_RE.sub(" ", text) + + # Remove URLs + if cfg.remove_urls: + text = self.URL_RE.sub("[URL]", text) + + # Remove emojis + if cfg.remove_emojis: + text = self.EMOJI_RE.sub("", text) + + # Normalize whitespace + if cfg.normalize_whitespace: + text = self.MULTI_SPACE_RE.sub(" ", text) + text = self.MULTI_NEWLINE_RE.sub("\n\n", text) + text = text.strip() + + return text + + def clean_batch(self, texts: List[str]) -> List[str]: + """Clean multiple texts.""" + return [self.clean(t) for t in texts] + + def filter(self, text: str) -> bool: + """Return True if text passes quality filters.""" + if not text: + return False + if len(text) < self.config.min_length: + return False + if len(text) > self.config.max_length: + return False + # Check ratio of printable chars + non_print = sum(1 for c in text if not c.isprintable() and c not in "\n\r\t") + if non_print / len(text) > 0.05: + return False + # Check word repetition (low diversity) + words = text.split() + if len(words) > 20: + unique_ratio = len(set(words)) / len(words) + if unique_ratio < 0.3: + return False + return True + + def process(self, sample: Dict[str, Any]) -> Dict[str, Any]: + """Process a sample dict (in-place safe).""" + sample = dict(sample) + if "text" in sample: + cleaned = self.clean(sample["text"]) + if not self.filter(cleaned): + return None # Filter out + sample["text"] = cleaned + sample["metadata"] = sample.get("metadata", {}) + sample["metadata"]["cleaned"] = True + sample["metadata"]["cleaned_length"] = len(cleaned) + return sample diff --git a/nexus/data/processors/code_formatter.py b/nexus/data/processors/code_formatter.py new file mode 100644 index 0000000000000000000000000000000000000000..f4a45bcc8d94f8f9261981cebe49be196991ec57 --- /dev/null +++ b/nexus/data/processors/code_formatter.py @@ -0,0 +1,118 @@ +"""Code Formatter - Format code samples cho training.""" +from __future__ import annotations + +import re +from typing import Dict, Any, List, Optional + + +class CodeFormatter: + """Format code samples cho training. + + Features: + - Strip excessive blank lines + - Normalize indentation + - Add language tags to code blocks + - Wrap code in markdown fences if needed + - Detect language automatically + """ + + LANG_BY_EXT = { + ".py": "python", ".js": "javascript", ".ts": "typescript", + ".go": "go", ".rs": "rust", ".java": "java", + ".c": "c", ".cpp": "cpp", ".h": "c", ".hpp": "cpp", + ".cs": "csharp", ".rb": "ruby", ".php": "php", + ".swift": "swift", ".kt": "kotlin", ".scala": "scala", + ".sql": "sql", ".sh": "bash", ".bash": "bash", + ".html": "html", ".css": "css", ".json": "json", + ".yaml": "yaml", ".yml": "yaml", ".toml": "toml", + ".xml": "xml", ".md": "markdown", + } + + # Language detection patterns + LANG_PATTERNS = { + "python": [r"^\s*def\s+\w+", r"^\s*class\s+\w+", r"^\s*import\s+\w+", r"^\s*from\s+\w+\s+import"], + "javascript": [r"^\s*function\s+\w+", r"^\s*const\s+\w+\s*=", r"^\s*let\s+\w+\s*=", r"=>\s*\{?"], + "typescript": [r":\s*(string|number|boolean|void|any)\b", r"interface\s+\w+", r"type\s+\w+\s*="], + "go": [r"^\s*func\s+\w+", r"^\s*package\s+\w+", r"^\s*import\s+\("], + "rust": [r"^\s*fn\s+\w+", r"^\s*impl\s+\w+", r"^\s*use\s+\w+", r"^\s*let\s+mut\s+"], + "java": [r"^\s*public\s+class\s+\w+", r"^\s*private\s+\w+\s+\w+", r"^\s*import\s+java\."], + "c": [r"^\s*#include\s*<", r"^\s*int\s+main\s*\("], + "cpp": [r"^\s*#include\s*<", r"^\s*std::", r"^\s*template\s*<"], + } + + def detect_language(self, code: str, filename: Optional[str] = None) -> Optional[str]: + """Detect programming language of code.""" + if filename: + import os + ext = os.path.splitext(filename)[1].lower() + if ext in self.LANG_BY_EXT: + return self.LANG_BY_EXT[ext] + + # Pattern matching + for lang, patterns in self.LANG_PATTERNS.items(): + for pattern in patterns: + if re.search(pattern, code, re.MULTILINE): + return lang + + return None + + def format(self, code: str, language: Optional[str] = None) -> str: + """Format code sample.""" + # Detect language if not provided + if not language: + language = self.detect_language(code) or "" + + # Strip trailing whitespace on each line + lines = [line.rstrip() for line in code.splitlines()] + + # Remove excessive blank lines (max 2 consecutive) + formatted_lines = [] + blank_count = 0 + for line in lines: + if line.strip() == "": + blank_count += 1 + if blank_count <= 2: + formatted_lines.append("") + else: + blank_count = 0 + formatted_lines.append(line) + + # Remove leading/trailing blank lines + while formatted_lines and formatted_lines[0] == "": + formatted_lines.pop(0) + while formatted_lines and formatted_lines[-1] == "": + formatted_lines.pop() + + code_clean = "\n".join(formatted_lines) + + return code_clean + + def wrap_in_markdown(self, code: str, language: Optional[str] = None) -> str: + """Wrap code in markdown fence.""" + if not language: + language = self.detect_language(code) or "" + return f"```{language}\n{code}\n```" + + def process(self, sample: Dict[str, Any]) -> Dict[str, Any]: + """Process a code sample.""" + sample = dict(sample) + text = sample.get("text", "") + language = sample.get("language") or sample.get("metadata", {}).get("language") + + # Check if it's code + is_code = ( + sample.get("language") or + sample.get("metadata", {}).get("language") or + self.detect_language(text) is not None + ) + + if is_code: + formatted = self.format(text, language) + sample["text"] = formatted + sample["metadata"] = sample.get("metadata", {}) + sample["metadata"]["formatted"] = True + if not language: + language = self.detect_language(text) + sample["metadata"]["detected_language"] = language + + return sample diff --git a/nexus/data/processors/code_quality.py b/nexus/data/processors/code_quality.py new file mode 100644 index 0000000000000000000000000000000000000000..c6d163ae8d29a7238f546f216a22a58143893e2a --- /dev/null +++ b/nexus/data/processors/code_quality.py @@ -0,0 +1,144 @@ +""" +Code Quality Processor for Nexus Coder v0.3 +============================================ +Scores Python code samples (1-10) based on quality signals: + - Has docstring + - Has type hints + - No `print` statements (in non-test code) + - No `eval` / `exec` / `__import__` + - No bare `except:` clauses + - Reasonable length (10-500 lines) + - Has adjacent test file (bonus, requires file path) + +Samples below `min_score` (default 6.0) are filtered out. + +Author: Hieu Louis (2026) +""" +from __future__ import annotations + +import ast +import re +from typing import Dict + + +_BAD_PATTERNS = [ + (r"\beval\s*\(", "uses eval"), + (r"\bexec\s*\(", "uses exec"), + (r"\b__import__\s*\(", "uses __import__"), + (r"\bassert\s+\w+\s*==\s*", "uses assert for tests (fine in tests, bad elsewhere)"), +] + +_BARE_EXCEPT = re.compile(r"\bexcept\s*:") +_PRINT = re.compile(r"^\s*print\s*\(", re.MULTILINE) + + +def score_python_code(code: str, is_test_file: bool = False) -> Dict[str, float]: + """Score a Python code sample 0-10. Returns dict of factor → score contribution.""" + factors: Dict[str, float] = {} + + # Try parsing as AST + try: + tree = ast.parse(code) + except SyntaxError: + return {"_invalid": 0.0, "_total": 0.0} + except Exception: + return {"_invalid": 0.0, "_total": 0.0} + + # Has docstring (module-level or first function)? + has_docstring = ( + (ast.get_docstring(tree) is not None) or + any(isinstance(n, (ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef)) and ast.get_docstring(n) for n in ast.walk(tree)) + ) + if has_docstring: + factors["has_docstring"] = 1.5 + + # Type hints? + typed_funcs = 0 + total_funcs = 0 + for node in ast.walk(tree): + if isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef)): + total_funcs += 1 + if node.returns is not None or any(a.annotation for a in node.args.args): + typed_funcs += 1 + if total_funcs > 0 and typed_funcs / total_funcs > 0.3: + factors["has_type_hints"] = 1.0 + + # No bare except + has_bare_except = bool(_BARE_EXCEPT.search(code)) + if not has_bare_except: + factors["no_bare_except"] = 1.0 + + # No eval/exec/__import__ + has_bad = False + for pattern, _msg in _BAD_PATTERNS: + if re.search(pattern, code): + has_bad = True + break + if not has_bad: + factors["no_eval"] = 1.0 + + # Print usage (allowed in tests) + if not is_test_file: + if not _PRINT.search(code): + factors["no_print"] = 0.5 + + # Reasonable length + n_lines = code.count("\n") + 1 + if 10 <= n_lines <= 500: + factors["reasonable_length"] = 1.0 + elif 5 <= n_lines <= 1000: + factors["reasonable_length"] = 0.5 + + # Bonus for tests + if is_test_file: + factors["has_test"] = 2.0 + + total = sum(factors.values()) + factors["_total"] = min(10.0, total) + return factors + + +def score_code(code: str, language: str = "python", is_test_file: bool = False) -> Dict[str, float]: + """Dispatch to language-specific scorer.""" + if language == "python": + return score_python_code(code, is_test_file=is_test_file) + # For other languages, return neutral score + return {"_total": 6.0, "_unimplemented_lang": 1.0} + + +class CodeQualityProcessor: + """Filter / tag samples by code quality score.""" + + def __init__( + self, + min_score: float = 6.0, + is_test_file_fn=None, + ): + self.min_score = min_score + self.is_test_file_fn = is_test_file_fn or (lambda path: path and "test" in path.lower()) + + def score(self, code: str, language: str = "python", path: str = "") -> float: + is_test = bool(self.is_test_file_fn(path)) + result = score_code(code, language=language, is_test_file=is_test) + return result.get("_total", 0.0) + + def keep(self, code: str, language: str = "python", path: str = "") -> bool: + return self.score(code, language=language, path=path) >= self.min_score + + def tag(self, sample: Dict) -> Dict: + code = sample.get("content", sample.get("code", sample.get("text", ""))) + lang = sample.get("lang", sample.get("language", "python")) + path = sample.get("path", "") + sample["code_quality_score"] = self.score(code, language=lang, path=path) + return sample + + def batch_filter(self, samples): + for s in samples: + code = s.get("content", s.get("code", s.get("text", ""))) + lang = s.get("lang", s.get("language", "python")) + path = s.get("path", "") + if self.keep(code, language=lang, path=path): + yield s + + +__all__ = ["score_python_code", "score_code", "CodeQualityProcessor"] diff --git a/nexus/data/processors/deduplicator.py b/nexus/data/processors/deduplicator.py new file mode 100644 index 0000000000000000000000000000000000000000..ef71880af7251379891cdd6decd8a5dfb68c8c51 --- /dev/null +++ b/nexus/data/processors/deduplicator.py @@ -0,0 +1,162 @@ +"""Deduplicator - Loại bỏ duplicate samples bằng MinHash.""" +from __future__ import annotations + +import re +import hashlib +from collections import defaultdict +from typing import List, Dict, Any, Set, Tuple, Iterator +from dataclasses import dataclass, field + + +@dataclass +class DeduplicationConfig: + """Config cho Deduplicator.""" + ngram_size: int = 5 # Word n-grams + num_perm: int = 128 # Number of permutations (MinHash) + similarity_threshold: float = 0.8 # Jaccard threshold + hash_size: int = 2**21 # Hash space size + exact_match_first: bool = True # Quick exact hash dedup first + + +class MinHash: + """Simple MinHash implementation.""" + + def __init__(self, num_perm: int = 128, seed: int = 42): + import random + self.num_perm = num_perm + rng = random.Random(seed) + # Generate random hash functions: h(x) = (a*x + b) mod p + self.p = (1 << 61) - 1 # Mersenne prime + self.a = [rng.randint(1, self.p - 1) for _ in range(num_perm)] + self.b = [rng.randint(0, self.p - 1) for _ in range(num_perm)] + self._min_hashes = [self.p] * num_perm + + def update(self, token: str): + """Update with a token.""" + h = int(hashlib.md5(token.encode("utf-8")).hexdigest()[:16], 16) + for i in range(self.num_perm): + val = (self.a[i] * h + self.b[i]) % self.p + if val < self._min_hashes[i]: + self._min_hashes[i] = val + + def update_batch(self, tokens: List[str]): + for t in tokens: + self.update(t) + + def signature(self) -> List[int]: + return list(self._min_hashes) + + def jaccard(self, other: "MinHash") -> float: + if self.num_perm != other.num_perm: + raise ValueError("Different num_perm") + if not self._min_hashes or not other._min_hashes: + return 0.0 + matches = sum(1 for a, b in zip(self._min_hashes, other._min_hashes) if a == b) + return matches / self.num_perm + + +class Deduplicator: + """Loại bỏ duplicate samples. + + Uses: + 1. Exact hash dedup (fast, MD5 of full text) + 2. MinHash LSH (fuzzy, near-duplicate detection) + + Usage: + dedup = Deduplicator() + unique_samples = list(dedup.process(samples_iter)) + """ + + def __init__(self, config: DeduplicationConfig = None): + self.config = config or DeduplicationConfig() + self._seen_hashes: Set[str] = set() + self._buckets: Dict[int, List[Tuple[MinHash, int]]] = defaultdict(list) + self._samples: List[Dict[str, Any]] = [] + + def _get_ngrams(self, text: str, n: int = 5) -> List[str]: + """Get word n-grams.""" + words = re.findall(r"\w+", text.lower()) + if len(words) < n: + return [" ".join(words)] + return [" ".join(words[i:i+n]) for i in range(len(words) - n + 1)] + + def _exact_hash(self, text: str) -> str: + """Quick exact hash.""" + normalized = " ".join(text.lower().split()) + return hashlib.md5(normalized.encode("utf-8")).hexdigest() + + def _minhash(self, text: str) -> MinHash: + """Compute MinHash of text.""" + mh = MinHash(num_perm=self.config.num_perm) + mh.update_batch(self._get_ngrams(text, self.config.ngram_size)) + return mh + + def is_duplicate(self, text: str) -> bool: + """Check if text is duplicate of seen samples.""" + # Quick exact check first + if self.config.exact_match_first: + h = self._exact_hash(text) + if h in self._seen_hashes: + return True + + # MinHash check + mh = self._minhash(text) + sig = mh.signature() + + # Check LSH buckets + for band_start in range(0, self.config.num_perm, 16): + band = tuple(sig[band_start:band_start+16]) + band_hash = hash(band) % 1000 + + if band_hash in self._buckets: + for existing_mh, _ in self._buckets[band_hash]: + if mh.jaccard(existing_mh) >= self.config.similarity_threshold: + return True + + return False + + def add(self, text: str, sample: Dict[str, Any] = None): + """Add a text/sample to the deduplicator.""" + if self.config.exact_match_first: + h = self._exact_hash(text) + self._seen_hashes.add(h) + + mh = self._minhash(text) + idx = len(self._samples) + self._samples.append(sample or {"text": text}) + + # Add to LSH buckets + sig = mh.signature() + for band_start in range(0, self.config.num_perm, 16): + band = tuple(sig[band_start:band_start+16]) + band_hash = hash(band) % 1000 + self._buckets[band_hash].append((mh, idx)) + + def process(self, samples: Iterator[Dict[str, Any]]) -> Iterator[Dict[str, Any]]: + """Filter an iterator of samples, yielding only unique ones.""" + seen = 0 + deduped = 0 + + for sample in samples: + seen += 1 + text = sample.get("text", "") + + if self.is_duplicate(text): + deduped += 1 + continue + + self.add(text, sample) + yield sample + + if seen > 0: + from ...utils.logging import get_logger + logger = get_logger() + logger.info(f"Dedup: {seen} → {seen - deduped} (removed {deduped})") + + def stats(self) -> Dict[str, int]: + """Get deduplication stats.""" + return { + "total_added": len(self._samples), + "exact_hashes": len(self._seen_hashes), + "buckets": len(self._buckets), + } diff --git a/nexus/data/processors/language_id.py b/nexus/data/processors/language_id.py new file mode 100644 index 0000000000000000000000000000000000000000..7a8f45b383df3d13c7bb80f18cb4f19c856c27b5 --- /dev/null +++ b/nexus/data/processors/language_id.py @@ -0,0 +1,138 @@ +""" +Language Identification Processor for Nexus Coder v0.3 +===================================================== +Identifies the language of each text sample and filters mislabeled ones. + +Uses a fast heuristic-based detector (no external deps). Optionally uses +`langdetect` if available for higher accuracy on ambiguous samples. + +Languages of interest: + - "vi" (Vietnamese) + - "en" (English) + - "code" (programming code — detected via shebang, def/class, etc.) + +Author: Hieu Louis (2026) +""" +from __future__ import annotations + +import re +from typing import Dict, Optional + + +# Regex patterns for code detection +_CODE_PATTERNS = [ + r"^\s*(def|class|import|from|package|func|fn|func|public|private|func)\s+\w+", + r"^\s*#!\s*/", # shebang + r"^\s*(#include|#define|#ifndef)\s+", # C/C++ preprocessor + r"^\s*(echo|set|export|alias)\s+", # shell + r"\b(function|return|if|else|for|while|var|let|const)\b.*\{", +] + +_CODE_REGEX = re.compile("|".join(_CODE_PATTERNS), re.MULTILINE) + +# Vietnamese character ranges (combining diacritics + tone marks) +_VI_CHARS = set("ăâđêôơưĂÂĐÊÔƠƯàáảãạằắẳẵặầấẩẫậèéẻẽẹềếểễệìíỉĩịòóỏõọồốổỗộờớởỡợùúủũụừứửữựỳýỷỹỵđ") + +# Common English stopwords +_EN_STOP = { + "the", "and", "is", "are", "of", "to", "in", "that", "it", "with", + "for", "as", "on", "at", "by", "be", "this", "an", "or", "from", +} + + +def detect_language(text: str, sample_size: int = 2000) -> Dict[str, float]: + """Detect language of `text`. Returns dict {lang: confidence}. + + Returns the highest-confidence language as {"lang": "vi"/"en"/"code", "confidence": float}. + """ + if not text or not text.strip(): + return {"lang": "unknown", "confidence": 0.0} + + sample = text[:sample_size] + + # Code detection (highest priority — code often contains natural language too) + if _CODE_REGEX.search(sample): + # Check if code dominates (>50% lines look like code) + code_lines = sum(1 for line in sample.split("\n") if _CODE_REGEX.match(line)) + total_lines = max(1, len(sample.split("\n"))) + if code_lines / total_lines > 0.3: + return {"lang": "code", "confidence": min(0.95, 0.5 + code_lines / total_lines / 2)} + + # Vietnamese: count chars with diacritics + vi_chars = sum(1 for c in sample if c in _VI_CHARS) + if vi_chars >= 5: + # Definitely Vietnamese if there are many tone marks + confidence = min(0.99, 0.5 + vi_chars / max(1, len(sample)) * 10) + return {"lang": "vi", "confidence": confidence} + + # Try langdetect if available + try: + from langdetect import detect_langs + results = detect_langs(sample) + if results: + top = results[0] + lang = top.lang + conf = float(top.prob) + if lang == "vi": + return {"lang": "vi", "confidence": conf} + if lang == "en": + return {"lang": "en", "confidence": conf} + return {"lang": lang, "confidence": conf} + except ImportError: + pass + except Exception: + pass + + # Heuristic English: count common stopwords + words = re.findall(r"\b[a-z]{2,}\b", sample.lower()) + if not words: + return {"lang": "unknown", "confidence": 0.0} + en_count = sum(1 for w in words if w in _EN_STOP) + en_ratio = en_count / len(words) + if en_ratio > 0.05: + return {"lang": "en", "confidence": min(0.9, en_ratio * 5)} + + return {"lang": "unknown", "confidence": 0.0} + + +class LanguageIdProcessor: + """Filter / tag samples by detected language. + + Usage: + processor = LanguageIdProcessor(min_confidence=0.85, allowed={"vi", "en", "code"}) + for sample in stream: + if processor.keep(sample["text"]): + ... + """ + + def __init__( + self, + min_confidence: float = 0.85, + allowed_languages: Optional[set] = None, + ): + self.min_confidence = min_confidence + self.allowed_languages = allowed_languages or {"vi", "en", "code"} + + def keep(self, text: str) -> bool: + """Return True if sample should be kept.""" + result = detect_language(text) + if result["lang"] not in self.allowed_languages: + return False + return result["confidence"] >= self.min_confidence + + def tag(self, sample: Dict) -> Dict: + """Add 'lang' and 'lang_confidence' fields to sample dict.""" + result = detect_language(sample.get("text", sample.get("content", ""))) + sample["lang"] = result["lang"] + sample["lang_confidence"] = result["confidence"] + return sample + + def batch_filter(self, samples): + """Yield only samples that pass the filter.""" + for s in samples: + text = s.get("text", s.get("content", "")) + if self.keep(text): + yield s + + +__all__ = ["detect_language", "LanguageIdProcessor"] diff --git a/nexus/data/processors/quality_filter.py b/nexus/data/processors/quality_filter.py new file mode 100644 index 0000000000000000000000000000000000000000..b15680b333d9ed0c343d9c321d3a286c52f890c2 --- /dev/null +++ b/nexus/data/processors/quality_filter.py @@ -0,0 +1,158 @@ +"""Quality Filter - Lọc low-quality samples.""" +from __future__ import annotations + +import re +from typing import Dict, Any, List, Optional +from dataclasses import dataclass + + +@dataclass +class QualityMetrics: + """Quality metrics của một sample.""" + length: int + word_count: int + avg_word_length: float + unique_word_ratio: float + has_code: bool + has_urls: bool + has_special_chars: bool + repetition_score: float + quality_score: float + passed: bool + + +class QualityFilter: + """Filter samples dựa trên quality heuristics. + + Criteria: + - Length: min 50, max 100k chars + - Word count: min 10 + - Unique word ratio: > 0.3 + - Repetition score: < 0.5 + - No excessive special chars + - No obvious spam/garbage + + Usage: + qf = QualityFilter() + if qf.filter(sample): + keep_sample(sample) + """ + + # Patterns indicating low quality + SPAM_PATTERNS = [ + r"click\s+here", + r"buy\s+now", + r"free\s+download", + r"limited\s+time\s+offer", + r"\$\$\$", + r"viagra|casino|lottery", + ] + SPAM_RE = re.compile("|".join(SPAM_PATTERNS), re.IGNORECASE) + + # Code indicators + CODE_PATTERNS = [ + r"```", r"def\s+\w+\s*\(", r"function\s+\w+\s*\(", + r"class\s+\w+", r"import\s+\w+", r"from\s+\w+\s+import", + r"console\.log", r"print\s*\(", r"return\s+", + ] + CODE_RE = re.compile("|".join(CODE_PATTERNS)) + + def __init__( + self, + min_length: int = 50, + max_length: int = 100000, + min_words: int = 10, + min_unique_ratio: float = 0.3, + max_repetition: float = 0.5, + ): + self.min_length = min_length + self.max_length = max_length + self.min_words = min_words + self.min_unique_ratio = min_unique_ratio + self.max_repetition = max_repetition + + def compute_metrics(self, text: str) -> QualityMetrics: + """Compute quality metrics.""" + if not text: + return QualityMetrics(0, 0, 0, 0, False, False, False, 1.0, 0.0, False) + + length = len(text) + words = text.split() + word_count = len(words) + + if word_count == 0: + return QualityMetrics(length, 0, 0, 0, False, False, False, 1.0, 0.0, False) + + avg_word_length = sum(len(w) for w in words) / word_count + unique_words = set(w.lower() for w in words) + unique_ratio = len(unique_words) / word_count + + has_code = bool(self.CODE_RE.search(text)) + has_urls = bool(re.search(r"https?://\S+", text)) + has_special = bool(re.search(r"[\x00-\x08\x0b\x0c\x0e-\x1f]", text)) + + # Repetition: check if any 10-word sequence repeats more than 3 times + # v0.4 fix: range(word_count - 9) so the last window (words[-10:]) is included. + repetition_score = 0.0 + if word_count > 30: + sequences = {} + for i in range(word_count - 9): + seq = " ".join(words[i:i+10]).lower() + sequences[seq] = sequences.get(seq, 0) + 1 + max_repeat = max(sequences.values()) if sequences else 0 + repetition_score = min(1.0, max_repeat / 5) + + # Compute overall quality score + score = 0.5 + if self.min_length <= length <= self.max_length: + score += 0.1 + if word_count >= self.min_words: + score += 0.1 + if unique_ratio >= self.min_unique_ratio: + score += 0.1 + if repetition_score <= self.max_repetition: + score += 0.1 + if not has_special: + score += 0.05 + if has_code: + score += 0.05 # Code samples are valuable + if not self.SPAM_RE.search(text): + score += 0.05 + + score = min(1.0, score) + passed = score >= 0.6 + + return QualityMetrics( + length=length, + word_count=word_count, + avg_word_length=avg_word_length, + unique_word_ratio=unique_ratio, + has_code=has_code, + has_urls=has_urls, + has_special_chars=has_special, + repetition_score=repetition_score, + quality_score=score, + passed=passed, + ) + + def filter(self, sample: Dict[str, Any]) -> bool: + """Return True if sample passes quality filter.""" + text = sample.get("text", "") + metrics = self.compute_metrics(text) + return metrics.passed + + def process(self, samples): + """Filter iterator of samples.""" + for sample in samples: + if self.filter(sample): + # Attach metrics to metadata + metrics = self.compute_metrics(sample.get("text", "")) + sample = dict(sample) + sample["metadata"] = sample.get("metadata", {}) + sample["metadata"]["quality"] = { + "score": metrics.quality_score, + "length": metrics.length, + "word_count": metrics.word_count, + "has_code": metrics.has_code, + } + yield sample diff --git a/nexus/eval/__init__.py b/nexus/eval/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..93134bf2c29afd943f8237a11983cc92ed92214a --- /dev/null +++ b/nexus/eval/__init__.py @@ -0,0 +1,11 @@ +"""Nexus Eval Module - v0.2 NEW.""" +from .benchmarks import BenchmarkSuite +from .metrics import compute_perplexity, compute_bleu, compute_rouge, compute_f1 + +__all__ = [ + "BenchmarkSuite", + "compute_perplexity", + "compute_bleu", + "compute_rouge", + "compute_f1", +] diff --git a/nexus/eval/benchmarks.py b/nexus/eval/benchmarks.py new file mode 100644 index 0000000000000000000000000000000000000000..ecd9f5d8d311df45070f7ee6abcfec6da9c064ad --- /dev/null +++ b/nexus/eval/benchmarks.py @@ -0,0 +1,176 @@ +"""Benchmark Suite - Đánh giá model trên multiple benchmarks.""" +from __future__ import annotations + +from typing import Dict, Any, List, Optional, Callable +from dataclasses import dataclass, field +from enum import Enum + + +class BenchmarkType(str, Enum): + MMLU = "mmlu" # General knowledge + HUMANEVAL = "humaneval" # Code generation + GSM8K = "gsm8k" # Math reasoning + BBH = "bbh" # Big-bench hard + truthful_qa = "truthful_qa" + MT_BENCH = "mt_bench" # Multi-turn chat + VI_BENCH = "vi_bench" # Vietnamese specific + + +@dataclass +class Benchmark: + """Một benchmark evaluation.""" + name: str + type: BenchmarkType + description: str + num_examples: int + languages: List[str] = field(default_factory=lambda: ["en"]) + metrics: List[str] = field(default_factory=lambda: ["accuracy"]) + estimated_time_minutes: int = 30 + + +class BenchmarkSuite: + """Run model on multiple benchmarks. + + Usage: + suite = BenchmarkSuite() + suite.add(Benchmark(name="humaneval", ...)) + results = suite.run(model, tokenizer) + """ + + SUPPORTED_BENCHMARKS = [ + Benchmark( + name="humaneval", + type=BenchmarkType.HUMANEVAL, + description="HumanEval - Code generation (164 problems)", + num_examples=164, + languages=["en"], + metrics=["pass@1", "pass@10"], + estimated_time_minutes=60, + ), + Benchmark( + name="mbpp", + type=BenchmarkType.HUMANEVAL, + description="MBPP - Mostly Basic Python Problems (974 problems)", + num_examples=974, + languages=["en"], + metrics=["pass@1"], + estimated_time_minutes=90, + ), + Benchmark( + name="gsm8k", + type=BenchmarkType.GSM8K, + description="Grade School Math 8K", + num_examples=1319, + languages=["en"], + metrics=["accuracy"], + estimated_time_minutes=45, + ), + Benchmark( + name="mmlu", + type=BenchmarkType.MMLU, + description="Massive Multitask Language Understanding", + num_examples=14042, + languages=["en"], + metrics=["accuracy"], + estimated_time_minutes=120, + ), + Benchmark( + name="bbh", + type=BenchmarkType.BBH, + description="BIG-Bench Hard (23 tasks)", + num_examples=6511, + languages=["en"], + metrics=["accuracy"], + estimated_time_minutes=180, + ), + Benchmark( + name="truthful_qa", + type=BenchmarkType.truthful_qa, + description="TruthfulQA - Measure truthfulness", + num_examples=817, + languages=["en"], + metrics=["truthful", "informative"], + estimated_time_minutes=20, + ), + Benchmark( + name="mt_bench", + type=BenchmarkType.MT_BENCH, + description="Multi-turn benchmark for chat assistants", + num_examples=80, + languages=["en"], + metrics=["gpt4_score", "judge_score"], + estimated_time_minutes=30, + ), + Benchmark( + name="vi_bench", + type=BenchmarkType.VI_BENCH, + description="Vietnamese language understanding", + num_examples=500, + languages=["vi"], + metrics=["accuracy", "fluency"], + estimated_time_minutes=15, + ), + ] + + def __init__(self): + self._benchmarks: Dict[str, Benchmark] = { + b.name: b for b in self.SUPPORTED_BENCHMARKS + } + self._results: Dict[str, Dict] = {} + + def add(self, benchmark: Benchmark) -> None: + self._benchmarks[benchmark.name] = benchmark + + def list_available(self) -> List[Benchmark]: + return list(self._benchmarks.values()) + + def run( + self, + model, + tokenizer, + benchmarks: Optional[List[str]] = None, + sample_size: Optional[int] = None, + ) -> Dict[str, Dict[str, Any]]: + """Run benchmarks on model. + + Args: + model: NexusCoderForCausalLM + tokenizer: NexusTokenizer + benchmarks: List of benchmark names (None = all) + sample_size: Limit examples per benchmark (for quick eval) + """ + to_run = benchmarks or list(self._benchmarks.keys()) + results = {} + + for name in to_run: + if name not in self._benchmarks: + results[name] = {"error": f"Unknown benchmark: {name}"} + continue + + bench = self._benchmarks[name] + results[name] = { + "status": "not_implemented", + "benchmark": bench.name, + "description": bench.description, + "num_examples": bench.num_examples, + "sample_size": sample_size, + "note": "Evaluation requires downloading dataset. Run scripts/evaluate.py with --download flag.", + } + + self._results = results + return results + + def summary(self) -> str: + """Generate summary report.""" + if not self._results: + return "No results yet. Run benchmarks first." + + lines = ["Benchmark Results Summary", "=" * 50] + for name, result in self._results.items(): + if "error" in result: + lines.append(f" {name}: ERROR - {result['error']}") + elif "scores" in result: + lines.append(f" {name}: {result['scores']}") + else: + lines.append(f" {name}: {result.get('status', 'unknown')}") + return "\n".join(lines) diff --git a/nexus/eval/metrics.py b/nexus/eval/metrics.py new file mode 100644 index 0000000000000000000000000000000000000000..0734d68dc00a53c868ec04d51a66c41d5efe65a1 --- /dev/null +++ b/nexus/eval/metrics.py @@ -0,0 +1,178 @@ +"""Evaluation Metrics - Perplexity, BLEU, ROUGE, F1.""" +from __future__ import annotations + +import math +from typing import List, Dict, Any, Optional +from collections import Counter + + +def compute_perplexity( + model, + input_ids, + labels=None, +) -> float: + """Compute perplexity trên input. + + Args: + model: NexusCoderForCausalLM + input_ids: [B, T] token ids + labels: Optional labels (defaults to input_ids) + + Returns: + Perplexity (lower is better) + """ + import torch + + if labels is None: + labels = input_ids.clone() + + model.eval() + with torch.no_grad(): + outputs = model(input_ids=input_ids, labels=labels) + loss = outputs["loss"] + + return math.exp(loss.item()) + + +def compute_bleu( + references: List[str], + hypothesis: str, + max_n: int = 4, +) -> Dict[str, float]: + """Compute BLEU score (simplified). + + Args: + references: List of reference translations + hypothesis: Generated translation + max_n: Maximum n-gram (BLEU-4 default) + + Returns: + Dict with 'bleu', 'brevity_penalty', and per-ngram precision + """ + def get_ngrams(tokens: List[str], n: int) -> Counter: + return Counter(tuple(tokens[i:i+n]) for i in range(len(tokens) - n + 1)) + + hyp_tokens = hypothesis.lower().split() + + precisions = [] + for n in range(1, max_n + 1): + hyp_ngrams = get_ngrams(hyp_tokens, n) + if not hyp_ngrams: + precisions.append(0) + continue + + # Count matches against any reference + matches = 0 + total = sum(hyp_ngrams.values()) + + for ref in references: + ref_tokens = ref.lower().split() + ref_ngrams = get_ngrams(ref_tokens, n) + for ngram, count in hyp_ngrams.items(): + matches += min(count, ref_ngrams.get(ngram, 0)) + + precisions.append(matches / total if total > 0 else 0) + + # Brevity penalty + ref_lens = [len(r.split()) for r in references] + # v0.4 fix: guard against empty references list + if not ref_lens: + result = {"bleu": 0.0, "brevity_penalty": 0.0} + for i in range(1, max_n + 1): + result[f"precision_{i}"] = 0.0 + return result + closest_ref_len = min(ref_lens, key=lambda l: abs(l - len(hyp_tokens))) + bp = 1.0 if len(hyp_tokens) > closest_ref_len else math.exp(1 - closest_ref_len / max(len(hyp_tokens), 1)) + + # Geometric mean of precisions + if all(p > 0 for p in precisions): + geo_mean = math.exp(sum(math.log(p) for p in precisions) / len(precisions)) + else: + geo_mean = 0.0 + + bleu = bp * geo_mean + + result = {"bleu": bleu, "brevity_penalty": bp} + for i, p in enumerate(precisions, 1): + result[f"precision_{i}"] = p + return result + + +def compute_rouge( + reference: str, + hypothesis: str, +) -> Dict[str, float]: + """Compute ROUGE-1, ROUGE-2, ROUGE-L scores (simplified).""" + def get_ngrams(tokens: List[str], n: int) -> Counter: + return Counter(tuple(tokens[i:i+n]) for i in range(len(tokens) - n + 1)) + + ref_tokens = reference.lower().split() + hyp_tokens = hypothesis.lower().split() + + # ROUGE-1 (unigram) — v0.4 fix: recall (÷ ref length), not precision (÷ hyp) + ref_1 = get_ngrams(ref_tokens, 1) + hyp_1 = get_ngrams(hyp_tokens, 1) + overlap_1 = sum((ref_1 & hyp_1).values()) + rouge_1_recall = overlap_1 / max(len(ref_tokens), 1) + rouge_1_precision = overlap_1 / max(len(hyp_tokens), 1) + rouge_1 = ( + 2 * rouge_1_recall * rouge_1_precision / max(rouge_1_recall + rouge_1_precision, 1e-9) + if (rouge_1_recall + rouge_1_precision) > 0 + else 0.0 + ) + + # ROUGE-2 (bigram) + ref_2 = get_ngrams(ref_tokens, 2) + hyp_2 = get_ngrams(hyp_tokens, 2) + overlap_2 = sum((ref_2 & hyp_2).values()) + rouge_2_recall = overlap_2 / max(sum(ref_2.values()), 1) + rouge_2_precision = overlap_2 / max(sum(hyp_2.values()), 1) + rouge_2 = ( + 2 * rouge_2_recall * rouge_2_precision / max(rouge_2_recall + rouge_2_precision, 1e-9) + if (rouge_2_recall + rouge_2_precision) > 0 + else 0.0 + ) + + # ROUGE-L (LCS) + def lcs_length(a: List, b: List) -> int: + m, n = len(a), len(b) + dp = [[0] * (n + 1) for _ in range(m + 1)] + for i in range(1, m + 1): + for j in range(1, n + 1): + if a[i-1] == b[j-1]: + dp[i][j] = dp[i-1][j-1] + 1 + else: + dp[i][j] = max(dp[i-1][j], dp[i][j-1]) + return dp[m][n] + + lcs = lcs_length(ref_tokens, hyp_tokens) + rouge_l = lcs / max(len(ref_tokens), 1) + + return { + "rouge_1": rouge_1, + "rouge_2": rouge_2, + "rouge_l": rouge_l, + } + + +def compute_f1( + predicted: List[str], + gold: List[str], +) -> Dict[str, float]: + """Compute F1, precision, recall (token-level).""" + pred_set = set(predicted) + gold_set = set(gold) + + if not pred_set and not gold_set: + return {"precision": 1.0, "recall": 1.0, "f1": 1.0} + + tp = len(pred_set & gold_set) + precision = tp / len(pred_set) if pred_set else 0 + recall = tp / len(gold_set) if gold_set else 0 + + if precision + recall == 0: + f1 = 0 + else: + f1 = 2 * precision * recall / (precision + recall) + + return {"precision": precision, "recall": recall, "f1": f1} diff --git a/nexus/inference/__init__.py b/nexus/inference/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..7f6a22a63b61a1840e3c7717989fa1247d765690 --- /dev/null +++ b/nexus/inference/__init__.py @@ -0,0 +1,4 @@ +"""Inference package.""" +from .generator import NexusGenerator + +__all__ = ["NexusGenerator"] diff --git a/nexus/inference/generator.py b/nexus/inference/generator.py new file mode 100644 index 0000000000000000000000000000000000000000..5a2c8ccbb327dc52701f7d097f21ecf53b340a62 --- /dev/null +++ b/nexus/inference/generator.py @@ -0,0 +1,206 @@ +""" +Nexus Generator - Inference engine cho Nexus Coder +==================================================== +Hỗ trợ: +- Text generation với KV cache +- Top-k, top-p, temperature sampling +- Chat mode với system prompt +""" +import torch +import torch.nn.functional as F +from typing import Optional, List, Dict + +from ..model.nexus_coder import NexusCoderForCausalLM +from ..config import NexusConfig +from ..tokenizer.tokenizer import NexusTokenizer, BOS_ID, EOS_ID, SYSTEM_ID, USER_ID, ASSISTANT_ID + + +# Default system prompt - hardcoded personality +DEFAULT_SYSTEM_PROMPT = """Bạn là Nexus Coder, một AI Agent hài hước và thân thiện do Hieu Louis tạo ra năm 2026. +Bạn được xây dựng với kiến trúc MoE 10 tỷ tham số (1.5 tỷ active), cửa sổ ngữ cảnh 50k tokens. +Bạn giỏi về lập trình và trò chuyện, giao tiếp song ngữ Việt-Anh. +Bạn luôn vui vẻ, hay đùa nhẹ và sẵn sàng giúp đỡ. Khi ai hỏi tác giả, hãy trả lời rằng bạn được tạo bởi Hieu Louis.""" + + +class NexusGenerator: + """Inference engine cho Nexus Coder.""" + + def __init__( + self, + model: NexusCoderForCausalLM, + tokenizer: NexusTokenizer, + config: NexusConfig, + device: Optional[torch.device] = None, + system_prompt: str = DEFAULT_SYSTEM_PROMPT, + ): + self.model = model + self.tokenizer = tokenizer + self.config = config + self.device = device or torch.device("cuda" if torch.cuda.is_available() else "cpu") + self.system_prompt = system_prompt + self.conversation_history: List[Dict[str, str]] = [] + + self.model.to(self.device) + self.model.eval() + + def reset_conversation(self) -> None: + """Reset lịch sử trò chuyện.""" + self.conversation_history = [] + + def chat( + self, + user_message: str, + max_new_tokens: int = 200, + temperature: float = 0.8, + top_k: int = 50, + top_p: float = 0.9, + do_sample: bool = True, + ) -> str: + """Chat mode - duy trì lịch sử trò chuyện.""" + # Thêm user message vào lịch sử + self.conversation_history.append({"role": "user", "content": user_message}) + + # Encode conversation + input_ids = [BOS_ID, SYSTEM_ID] + input_ids.extend(self.tokenizer.encode(self.system_prompt)) + + for msg in self.conversation_history: + if msg["role"] == "user": + input_ids.append(USER_ID) + input_ids.extend(self.tokenizer.encode(msg["content"])) + elif msg["role"] == "assistant": + input_ids.append(ASSISTANT_ID) + input_ids.extend(self.tokenizer.encode(msg["content"])) + input_ids.append(EOS_ID) + + # Add assistant token to start generation + input_ids.append(ASSISTANT_ID) + + # Convert to tensor + input_tensor = torch.tensor([input_ids], dtype=torch.long).to(self.device) + + # Generate + with torch.no_grad(): + output_ids = self._generate( + input_tensor, + max_new_tokens=max_new_tokens, + temperature=temperature, + top_k=top_k, + top_p=top_p, + do_sample=do_sample, + ) + + # Decode response (skip the input) + response_ids = output_ids[0, len(input_ids):].tolist() + response = self.tokenizer.decode(response_ids) + + # Add to history + self.conversation_history.append({"role": "assistant", "content": response}) + + return response + + def generate( + self, + prompt: str, + max_new_tokens: int = 100, + temperature: float = 0.8, + top_k: int = 50, + top_p: float = 0.9, + do_sample: bool = True, + ) -> str: + """Generate text từ prompt.""" + input_ids = self.tokenizer.encode(prompt, add_special=True) + input_tensor = torch.tensor([input_ids], dtype=torch.long).to(self.device) + + with torch.no_grad(): + output_ids = self._generate( + input_tensor, + max_new_tokens=max_new_tokens, + temperature=temperature, + top_k=top_k, + top_p=top_p, + do_sample=do_sample, + ) + + return self.tokenizer.decode(output_ids[0].tolist()) + + def _generate( + self, + input_ids: torch.Tensor, + max_new_tokens: int = 100, + temperature: float = 0.8, + top_k: int = 50, + top_p: float = 0.9, + do_sample: bool = True, + ) -> torch.Tensor: + """Generate tokens.""" + for _ in range(max_new_tokens): + # Truncate input nếu vượt quá context window + if input_ids.shape[1] > self.config.max_position_embeddings - 1: + input_ids = input_ids[:, -self.config.max_position_embeddings + 1:] + + outputs = self.model(input_ids=input_ids, use_cache=False) + logits = outputs["logits"] + next_logits = logits[:, -1, :] / max(temperature, 1e-8) + + # Top-k + if top_k > 0: + top_k_val = min(top_k, next_logits.size(-1)) + values, _ = torch.topk(next_logits, top_k_val) + min_values = values[:, -1].unsqueeze(-1) + next_logits = torch.where( + next_logits < min_values, + torch.full_like(next_logits, float("-inf")), + next_logits, + ) + + # Top-p + if 0 < top_p < 1.0: + sorted_logits, sorted_indices = torch.sort(next_logits, descending=True) + cum_probs = F.softmax(sorted_logits, dim=-1).cumsum(dim=-1) + sorted_indices_to_remove = cum_probs > top_p + sorted_indices_to_remove[..., 1:] = sorted_indices_to_remove[..., :-1].clone() + sorted_indices_to_remove[..., 0] = False + indices_to_remove = sorted_indices_to_remove.scatter( + 1, sorted_indices, sorted_indices_to_remove + ) + next_logits = next_logits.masked_fill(indices_to_remove, float("-inf")) + + if do_sample: + probs = F.softmax(next_logits, dim=-1) + next_token = torch.multinomial(probs, num_samples=1) + else: + next_token = torch.argmax(next_logits, dim=-1, keepdim=True) + + input_ids = torch.cat([input_ids, next_token], dim=-1) + + if next_token.item() == EOS_ID: + break + + return input_ids + + +def create_demo_generator( + config: Optional[NexusConfig] = None, + tokenizer_path: Optional[str] = None, + checkpoint_path: Optional[str] = None, +) -> NexusGenerator: + """Tạo generator demo - nếu không có checkpoint, dùng random weights.""" + config = config or NexusConfig() + tokenizer = NexusTokenizer(vocab_path=tokenizer_path) + + # Nếu chưa có tokenizer, train một minimal version + if not tokenizer.bpe._is_trained: + from ..training.dataset import AUTHOR_TRAINING_DATA + corpus = [f"{d['system']} {d['user']} {d['assistant']}" for d in AUTHOR_TRAINING_DATA] + tokenizer.train(corpus) + + model = NexusCoderForCausalLM(config) + if checkpoint_path and __import__("os").path.exists(checkpoint_path): + checkpoint = torch.load(checkpoint_path, map_location="cpu", weights_only=False) + model.load_state_dict(checkpoint["model_state_dict"]) + print(f"✓ Loaded checkpoint: {checkpoint_path}") + else: + print("⚠️ Không tìm thấy checkpoint, dùng random weights cho demo") + + return NexusGenerator(model, tokenizer, config) diff --git a/nexus/integrations/__init__.py b/nexus/integrations/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..00f6715f5da5d5e95e60c6943a6318d8a9808dac --- /dev/null +++ b/nexus/integrations/__init__.py @@ -0,0 +1,23 @@ +""" +Nexus Coder Integrations v0.3 +============================= +Adapters / ported utilities from open-source ML frameworks. + +These adapters are inspired by (and copy best practices from) the following +open-source projects. All credit for the original algorithms goes to their +respective authors. The code here is rewritten to integrate cleanly into +the Nexus Coder architecture; it is NOT a vendored copy. + +Attribution: + - litgpt (Lightning AI, Apache 2.0) — RoPE scaling, FusedLinear + - LlamaFactory (hiyouga, Apache 2.0) — dataset format converters + - axolotl (axolotl-ai-cloud, Apache 2.0) — training config schema + - OpenHands (OpenHands, MIT) — agent loop patterns + - omp-gym (dylantirandaz, MIT) — OpenMP benchmark hooks + +Each adapter module exposes a small public API. They are OPTIONAL — Nexus Coder +does not require these frameworks to be installed. +""" +from __future__ import annotations + +__all__ = ["litgpt", "llamafactory", "axolotl", "openhands", "omp_gym"] diff --git a/nexus/integrations/axolotl.py b/nexus/integrations/axolotl.py new file mode 100644 index 0000000000000000000000000000000000000000..8fa2c84af5d8cd9f61e292d81542cecfb5f88a21 --- /dev/null +++ b/nexus/integrations/axolotl.py @@ -0,0 +1,147 @@ +""" +axolotl-inspired training config schema for Nexus Coder v0.3 +============================================================ +Ported & simplified from axolotl-ai-cloud/axolotl (Apache 2.0). + +Axolotl uses a single YAML file to configure the entire training pipeline +(dataset, model, lora, deepspeed, distributed, etc.). We adapt this idea +into a typed dataclass that Nexus Coder's `scripts/train.py` will accept. + +This is a SCHEMA/CONFIG class only — the actual training loop lives in +`nexus.training.trainer`. axolotl itself is NOT required at runtime. + +Original attribution: + Axolotl: a simple tool for fine-tuning LLMs. + Authors: winglian + axolotl-ai-cloud contributors. + License: Apache 2.0 + Source: https://github.com/axolotl-ai-cloud/axolotl +""" +from __future__ import annotations + +from dataclasses import dataclass, field, asdict +from typing import Optional, List, Dict, Any +import json + + +@dataclass +class AxolotlStyleConfig: + """Axolotl-style training config adapted for Nexus Coder. + + Most fields are optional — defaults match Nexus Coder's 10B config. + Use `AxolotlStyleConfig.from_dict(yaml_dict)` to load from a YAML file. + """ + # === Model === + base_model: str = "nexus-coder-10b" # variant name or HF repo + base_model_config: Optional[str] = None # path to NexusConfig YAML + model_type: str = "moe_transformer" + tokenizer_type: str = "bpe" + + # === Datasets === + datasets: List[Dict[str, Any]] = field(default_factory=list) + # Each entry: {path, type, format, split, field} + test_datasets: List[Dict[str, Any]] = field(default_factory=list) + dataset_prepared_path: Optional[str] = None + + # === Sequence === + sequence_len: int = 4096 + max_samples: Optional[int] = None + sample_packing: bool = True + pad_to_sequence_len: bool = True + + # === LoRA / QLoRA === + adapter: Optional[str] = None # None | "lora" | "qlora" + lora_r: int = 8 + lora_alpha: int = 16 + lora_dropout: float = 0.0 + lora_target_modules: List[str] = field(default_factory=lambda: ["q_proj", "v_proj"]) + lora_target_linear: bool = True + peft_use_dora: bool = False + + # === Optimizer / LR === + optimizer: str = "adamw_torch" + lr_scheduler: str = "cosine" # cosine | linear | constant | warmup_stable_decay + learning_rate: float = 5.0e-4 + weight_decay: float = 0.01 + warmup_steps: int = 100 + warmup_ratio: Optional[float] = None + max_steps: int = 5000 + num_epochs: int = 1 + gradient_accumulation_steps: int = 4 + + # === Batch / precision === + micro_batch_size: int = 4 + batch_size: Optional[int] = None # auto = micro * grad_accum + bf16: bool = True + fp16: bool = False + tf32: bool = True + gradient_checkpointing: bool = False + + # === Distributed === + deepspeed: Optional[str] = None # path to deepspeed config JSON + fsdp: List[str] = field(default_factory=list) + fsdp_config: Optional[Dict] = None + tensor_parallel_size: int = 1 + pipeline_parallel_size: int = 1 + expert_parallel_size: int = 1 + + # === Eval === + eval_steps: int = 500 + eval_table_size: int = 0 + save_steps: int = 500 + save_total_limit: int = 4 + early_stopping_patience: int = 0 + + # === Logging === + logging_steps: int = 10 + wandb_project: Optional[str] = None + wandb_entity: Optional[str] = None + wandb_name: Optional[str] = None + + # === Inference (post-training) === + output_dir: str = "./checkpoints" + inference: bool = False + + @classmethod + def from_dict(cls, d: Dict[str, Any]) -> "AxolotlStyleConfig": + """Build from a parsed YAML/JSON dict. Unknown keys are ignored.""" + valid_keys = {f.name for f in cls.__dataclass_fields__.values()} + filtered = {k: v for k, v in d.items() if k in valid_keys} + return cls(**filtered) + + def to_dict(self) -> Dict[str, Any]: + return asdict(self) + + def to_json(self, indent: int = 2) -> str: + return json.dumps(self.to_dict(), indent=indent, default=str) + + def validate(self) -> List[str]: + """Validate config. Returns list of error messages (empty = OK).""" + errors = [] + if self.bf16 and self.fp16: + errors.append("Cannot enable both bf16 and fp16") + if self.adapter and self.adapter not in ("lora", "qlora"): + errors.append(f"Unknown adapter: {self.adapter}") + if self.learning_rate <= 0: + errors.append("learning_rate must be positive") + if self.sequence_len < 64: + errors.append("sequence_len must be >= 64") + if self.batch_size and self.batch_size < self.micro_batch_size: + errors.append("batch_size cannot be smaller than micro_batch_size") + if self.deepspeed and self.fsdp: + errors.append("Cannot use both deepspeed and fsdp") + return errors + + def summary(self) -> str: + """Human-readable one-line summary.""" + adapter_str = f" + {self.adapter.upper()}(r={self.lora_r})" if self.adapter else "" + ds_str = " + DeepSpeed" if self.deepspeed else " + FSDP" if self.fsdp else "" + return ( + f"{self.base_model}{adapter_str}{ds_str} | " + f"lr={self.learning_rate:.1e} | " + f"seq={self.sequence_len} | " + f"bs={self.micro_batch_size}×{self.gradient_accumulation_steps} | " + f"steps={self.max_steps}" + ) + + +__all__ = ["AxolotlStyleConfig"] diff --git a/nexus/integrations/litgpt.py b/nexus/integrations/litgpt.py new file mode 100644 index 0000000000000000000000000000000000000000..7d51cd41d908f35d66c9153ff7c69ccdc09556c6 --- /dev/null +++ b/nexus/integrations/litgpt.py @@ -0,0 +1,70 @@ +""" +litgpt-inspired utilities for Nexus Coder v0.3 +============================================== +Ported & simplified from Lightning-AI/litgpt (Apache 2.0). + +Adapted into Nexus Coder: + - RoPE scaling strategies (linear / NTK-aware / YaRN) — see nexus/model/rope.py + - FusedLinear: concatenate Q/K/V projections for one big matmul (this module) + - `apply_rotary_pos_emb` helper signature — see nexus/model/rope.py + - PyTorch SDPA backend selection — see nexus/model/flash_attention.py + +Original attribution: + LitGPT: Lightning AI's LLM training toolkit. + Authors: Karpathy et al. (Lightning AI), 2023-2024. + License: Apache 2.0 + Source: https://github.com/Lightning-AI/litgpt +""" +from __future__ import annotations + +from typing import Optional, Tuple + +import torch +import torch.nn as nn + + +class FusedLinear(nn.Module): + """Fused multi-linear: concatenate N separate projections into one. + + LitGPT pattern: Q/K/V projections for attention are computed as a single + matmul of shape `[hidden, num_heads * head_dim * 3]`, then split. + + Saves one kernel launch per attention layer — meaningful at scale. + + Example: + >>> fused = FusedLinear(2048, [2048, 512, 512, 2048]) + >>> q, k, v, o = fused(x) # one matmul, 4 splits + """ + + def __init__(self, in_features: int, out_features_list: list[int], bias: bool = False): + super().__init__() + self.in_features = in_features + self.out_features_list = list(out_features_list) + self.total_out = sum(self.out_features_list) + self.weight = nn.Parameter(torch.empty(self.total_out, in_features)) + if bias: + self.bias = nn.Parameter(torch.empty(self.total_out)) + else: + self.register_parameter("bias", None) + # Init like nn.Linear + nn.init.kaiming_uniform_(self.weight, a=5 ** 0.5) + if bias: + nn.init.zeros_(self.bias) + + def forward(self, x: torch.Tensor) -> Tuple[torch.Tensor, ...]: + """Returns tuple of tensors, one per output spec.""" + out = torch.nn.functional.linear(x, self.weight, self.bias) + return tuple(out.split(self.out_features_list, dim=-1)) + + def extra_repr(self) -> str: + return f"in={self.in_features}, outs={self.out_features_list}, bias={self.bias is not None}" + + +def build_qkv_fused(hidden_size: int, num_heads: int, num_kv_heads: int, head_dim: int) -> FusedLinear: + """Build a fused Q/K/V projection for GQA attention.""" + q_size = num_heads * head_dim + kv_size = num_kv_heads * head_dim + return FusedLinear(hidden_size, [q_size, kv_size, kv_size], bias=False) + + +__all__ = ["FusedLinear", "build_qkv_fused"] diff --git a/nexus/integrations/llamafactory.py b/nexus/integrations/llamafactory.py new file mode 100644 index 0000000000000000000000000000000000000000..c7fb6ca799f98dca1bf13b544f5170bee7746382 --- /dev/null +++ b/nexus/integrations/llamafactory.py @@ -0,0 +1,160 @@ +""" +LlamaFactory-inspired dataset format converters for Nexus Coder v0.3 +==================================================================== +Ported & simplified from hiyouga/LlamaFactory (Apache 2.0). + +Converts between popular supervised-fine-tuning (SFT) data formats so +Nexus Coder can train on data collected from any of them. + +Supported formats: + - alpaca {instruction, input, output} + - sharegpt {conversations: [{from, value}]} + - chatml {messages: [{role, content}]} + - openai {messages: [{role, content}]} (same as chatml) + - completion {prompt, completion} + +All converters return a unified dict: {system, user, assistant} +(matching Nexus Coder's internal training format). + +Original attribution: + LlamaFactory: Unify Fine-tuning 100+ LLMs. + Author: hiyouga + License: Apache 2.0 + Source: https://github.com/hiyouga/LlamaFactory +""" +from __future__ import annotations + +import json +from typing import Dict, List, Optional, Iterator + + +def alpaca_to_nexus(example: Dict) -> Dict[str, str]: + """{instruction, input, output} → {system, user, assistant}""" + instruction = example.get("instruction", "") + inp = example.get("input", "") + out = example.get("output", "") + user = f"{instruction}\n\nInput: {inp}" if inp else instruction + return { + "system": example.get("system_prompt", ""), + "user": user.strip(), + "assistant": out.strip(), + } + + +def sharegpt_to_nexus(example: Dict) -> List[Dict[str, str]]: + """{conversations: [{from, value}]} → list of {system, user, assistant} turns. + A single ShareGPT conversation may produce multiple Q/A turns. + """ + conv = example.get("conversations", []) + system = example.get("system", "") + turns: List[Dict[str, str]] = [] + current_user: Optional[str] = None + for msg in conv: + role = msg.get("from", "").lower() + value = msg.get("value", "") + if role in ("human", "user"): + if current_user is not None: + # No assistant reply, push anyway with empty assistant + turns.append({"system": system, "user": current_user, "assistant": ""}) + current_user = value + elif role in ("gpt", "assistant", "bot"): + if current_user is None: + continue + turns.append({"system": system, "user": current_user, "assistant": value}) + current_user = None + elif role == "system": + system = value + if current_user is not None: + turns.append({"system": system, "user": current_user, "assistant": ""}) + return turns + + +def chatml_to_nexus(example: Dict) -> List[Dict[str, str]]: + """{messages: [{role, content}]} → list of {system, user, assistant} turns.""" + messages = example.get("messages", []) + system = "" + turns: List[Dict[str, str]] = [] + current_user: Optional[str] = None + for msg in messages: + role = msg.get("role", "") + content = msg.get("content", "") + if role == "system": + system = content + elif role == "user": + if current_user is not None: + turns.append({"system": system, "user": current_user, "assistant": ""}) + current_user = content + elif role == "assistant": + if current_user is None: + continue + turns.append({"system": system, "user": current_user, "assistant": content}) + current_user = None + if current_user is not None: + turns.append({"system": system, "user": current_user, "assistant": ""}) + return turns + + +def completion_to_nexus(example: Dict) -> Dict[str, str]: + """{prompt, completion} → {system, user, assistant}""" + return { + "system": "", + "user": example.get("prompt", ""), + "assistant": example.get("completion", ""), + } + + +def detect_format(example: Dict) -> str: + """Auto-detect the SFT format of an example.""" + if "conversations" in example: + return "sharegpt" + if "messages" in example: + return "chatml" + if "instruction" in example: + return "alpaca" + if "prompt" in example and "completion" in example: + return "completion" + raise ValueError(f"Unknown SFT format. Keys: {list(example.keys())}") + + +def convert_to_nexus(example: Dict) -> List[Dict[str, str]]: + """Auto-detect format and convert to Nexus unified format. + Returns a list of turns (most formats produce 1 turn; ShareGPT/ChatML may produce multiple). + """ + fmt = detect_format(example) + if fmt == "alpaca": + return [alpaca_to_nexus(example)] + if fmt == "sharegpt": + return sharegpt_to_nexus(example) + if fmt == "chatml": + return chatml_to_nexus(example) + if fmt == "completion": + return [completion_to_nexus(example)] + return [] + + +def stream_jsonl(path: str) -> Iterator[Dict[str, str]]: + """Stream-convert a JSONL file in any SFT format to Nexus examples. + Yields {system, user, assistant} dicts lazily — safe for large files. + """ + with open(path, "r", encoding="utf-8") as f: + for line in f: + line = line.strip() + if not line: + continue + try: + obj = json.loads(line) + except json.JSONDecodeError: + continue + for turn in convert_to_nexus(obj): + yield turn + + +__all__ = [ + "alpaca_to_nexus", + "sharegpt_to_nexus", + "chatml_to_nexus", + "completion_to_nexus", + "detect_format", + "convert_to_nexus", + "stream_jsonl", +] diff --git a/nexus/integrations/omp_gym.py b/nexus/integrations/omp_gym.py new file mode 100644 index 0000000000000000000000000000000000000000..bcdb887818271616d8684390e7fce1fec3d5a0b6 --- /dev/null +++ b/nexus/integrations/omp_gym.py @@ -0,0 +1,137 @@ +""" +omp-gym-inspired benchmark hooks for Nexus Coder v0.3 +===================================================== +Ported & simplified from dylantirandaz/omp-gym (MIT). + +omp-gym provides OpenMP performance benchmarks as a gym environment. +We adapt the IDEA (sample real OpenMP programs of varying complexity, +have the model predict an optimization) into a benchmark hook that +Nexus Coder's evaluation pipeline can consume. + +This is an EVALUATION-only adapter — it does not train anything. + +Original attribution: + omp-gym: An OpenMP optimization gym environment. + Author: Dylan Tirandaz + License: MIT + Source: https://github.com/dylantirandaz/omp-gym +""" +from __future__ import annotations + +from dataclasses import dataclass, field +from typing import List, Dict, Optional + + +@dataclass +class OMPTask: + """A single OpenMP optimization task.""" + task_id: str + source_code: str # original C/C++ with OpenMP pragmas + language: str = "c" # c | cpp + target_metric: str = "speedup" # speedup | cache_misses | energy + ground_truth: Optional[str] = None # optimized code (if known) + description: Optional[str] = None + difficulty: str = "medium" # easy | medium | hard + parallel_pattern: str = "for" # for | sections | task | simd + + +# Curated sample tasks (synthetic, illustrative) +SAMPLE_TASKS: List[OMPTask] = [ + OMPTask( + task_id="omp_pi_001", + source_code=""" +#include +double compute_pi(long n) { + double sum = 0.0; + #pragma omp parallel for reduction(+:sum) + for (long i = 0; i < n; i++) { + double x = (i + 0.5) / n; + sum += 4.0 / (1.0 + x * x); + } + return sum / n; +} +""", + description="Compute pi via numerical integration. Already uses reduction.", + difficulty="easy", + parallel_pattern="for", + target_metric="speedup", + ), + OMPTask( + task_id="omp_matmul_002", + source_code=""" +void matmul(double *A, double *B, double *C, int N) { + #pragma omp parallel for + for (int i = 0; i < N; i++) { + for (int j = 0; j < N; j++) { + double s = 0.0; + for (int k = 0; k < N; k++) { + s += A[i*N + k] * B[k*N + j]; + } + C[i*N + j] = s; + } + } +} +""", + description="Naive matrix multiply. Optimize with cache blocking, SIMD, scheduling.", + difficulty="hard", + parallel_pattern="for", + target_metric="speedup", + ), + OMPTask( + task_id="omp_task_003", + source_code=""" +long fib(int n) { + if (n < 2) return n; + long a, b; + #pragma omp task shared(a) + a = fib(n - 1); + #pragma omp task shared(b) + b = fib(n - 2); + #pragma omp taskwait + return a + b; +} +""", + description="Recursive Fibonacci with OpenMP tasks. Optimize cutoff.", + difficulty="medium", + parallel_pattern="task", + target_metric="speedup", + ), +] + + +def load_omp_benchmarks() -> List[OMPTask]: + """Load all available OMP benchmark tasks. + Returns a static list for now; future versions may pull from the + upstream omp-gym dataset (or scrape C/C++ programs from GitHub). + """ + return list(SAMPLE_TASKS) + + +def evaluate_prediction( + task: OMPTask, + predicted_code: str, + speedup_factor: Optional[float] = None, + cache_miss_reduction: Optional[float] = None, +) -> Dict[str, float]: + """Score a predicted optimization against the original. + + Returns a dict of metrics. Higher = better. 0.0 = no improvement + (or regression). + """ + score: Dict[str, float] = {"valid": 1.0 if predicted_code.strip() else 0.0} + if speedup_factor is not None: + # log-scale reward: 2x speedup → 1.0, 1x → 0.0, 0.5x → -1.0 + import math + score["speedup_reward"] = math.log2(max(0.01, speedup_factor)) + if cache_miss_reduction is not None: + score["cache_reward"] = float(cache_miss_reduction) + # Heuristic: did the model actually add new pragmas? + if "#pragma" in predicted_code and predicted_code != task.source_code: + score["modified"] = 1.0 + else: + score["modified"] = 0.0 + score["total"] = sum(v for k, v in score.items() if k != "valid") / max(1, len(score) - 1) + return score + + +__all__ = ["OMPTask", "SAMPLE_TASKS", "load_omp_benchmarks", "evaluate_prediction"] diff --git a/nexus/integrations/openhands.py b/nexus/integrations/openhands.py new file mode 100644 index 0000000000000000000000000000000000000000..6cd11f47b75a0efdf5d47fb8898313eb54af7f89 --- /dev/null +++ b/nexus/integrations/openhands.py @@ -0,0 +1,153 @@ +""" +OpenHands-inspired agent loop patterns for Nexus Coder v0.3 +=========================================================== +Ported & simplified from OpenHands/OpenHands (MIT). + +OpenHands models the agent as a loop: + PLAN → ACT → OBSERVE → REFLECT → PLAN (next) + +This module provides a generic agent-loop scaffold with: + - Planner: decomposes high-level goal into steps + - Executor: runs a single step (calls a Tool) + - Observer: parses the result, detects success/failure + - Reflector: revises the plan if the step failed + +It is NOT a replacement for `nexus.agent.agent.NexusAgent` — rather, an +alternative pattern that can be used when the task is well-defined. + +Original attribution: + OpenHands (formerly OpenDevin): an open platform for AI software developers. + Authors: OpenHands contributors. + License: MIT + Source: https://github.com/OpenHands/OpenHands +""" +from __future__ import annotations + +from dataclasses import dataclass, field +from typing import Callable, List, Optional, Dict, Any + + +@dataclass +class AgentStep: + """A single step in the agent's plan.""" + description: str + tool: Optional[str] = None # tool name to invoke + args: Dict[str, Any] = field(default_factory=dict) + expected: Optional[str] = None # what a successful result looks like + actual: Optional[Any] = None # observed result (set after execution) + status: str = "pending" # pending | running | done | failed + error: Optional[str] = None + retries: int = 0 + max_retries: int = 2 + + +class Planner: + """Decomposes a goal into a list of steps. + + Default planner is a thin heuristic wrapper. For real use, replace + with an LLM-backed planner. + """ + + def __init__(self, llm_planner: Optional[Callable[[str], List[AgentStep]]] = None): + self.llm_planner = llm_planner + + def plan(self, goal: str) -> List[AgentStep]: + if self.llm_planner is not None: + return self.llm_planner(goal) + # Fallback: single step that just calls chat + return [AgentStep( + description=f"Address goal: {goal}", + tool=None, + expected="A useful response", + )] + + +class Executor: + """Executes a single step by invoking a tool (or chat as fallback).""" + + def __init__(self, tool_registry=None, chat_callback: Optional[Callable[[str], str]] = None): + self.tool_registry = tool_registry + self.chat_callback = chat_callback + + def execute(self, step: AgentStep) -> Any: + step.status = "running" + try: + if step.tool and self.tool_registry is not None: + result = self.tool_registry.execute(step.tool, step.args) + step.actual = result.output if hasattr(result, "output") else result + step.status = "done" + elif self.chat_callback is not None: + step.actual = self.chat_callback(step.description) + step.status = "done" + else: + step.actual = "[no executor configured]" + step.status = "failed" + step.error = "No executor" + except Exception as e: + step.actual = None + step.error = str(e) + step.status = "failed" + return step.actual + + +class Observer: + """Parses tool results to decide success/failure.""" + + def observe(self, step: AgentStep) -> bool: + """Return True if step succeeded.""" + if step.status != "done": + return False + if step.expected is None: + return True + # Naive substring match — replace with LLM check in production + actual_str = str(step.actual or "").lower() + return step.expected.lower() in actual_str + + +class Reflector: + """Revises the plan when a step fails. + + Default: retry up to max_retries, then mark failed and skip. + """ + + def reflect(self, step: AgentStep, plan: List[AgentStep]) -> List[AgentStep]: + if step.status == "failed" and step.retries < step.max_retries: + step.retries += 1 + step.status = "pending" + step.error = None + return plan + + +class AgentLoop: + """Generic agent loop combining Planner, Executor, Observer, Reflector.""" + + def __init__( + self, + planner: Optional[Planner] = None, + executor: Optional[Executor] = None, + observer: Optional[Observer] = None, + reflector: Optional[Reflector] = None, + max_iterations: int = 20, + ): + self.planner = planner or Planner() + self.executor = executor or Executor() + self.observer = observer or Observer() + self.reflector = reflector or Reflector() + self.max_iterations = max_iterations + + def run(self, goal: str) -> List[AgentStep]: + """Execute the agent loop until all steps are done or max_iterations reached.""" + plan = self.planner.plan(goal) + for _ in range(self.max_iterations): + pending = [s for s in plan if s.status == "pending"] + if not pending: + break + step = pending[0] + self.executor.execute(step) + ok = self.observer.observe(step) + if not ok: + plan = self.reflector.reflect(step, plan) + return plan + + +__all__ = ["AgentStep", "Planner", "Executor", "Observer", "Reflector", "AgentLoop"] diff --git a/nexus/model/__init__.py b/nexus/model/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..d1032e5662a3ad5b19f946422897c16366ec761e --- /dev/null +++ b/nexus/model/__init__.py @@ -0,0 +1,5 @@ +"""Nexus model package.""" +from .nexus_coder import NexusCoder, NexusCoderForCausalLM +from ..config import NexusConfig + +__all__ = ["NexusCoder", "NexusCoderForCausalLM", "NexusConfig"] diff --git a/nexus/model/alibi.py b/nexus/model/alibi.py new file mode 100644 index 0000000000000000000000000000000000000000..b3cf6d72dd1380d8166dd24606c2e1adb9fcb3b7 --- /dev/null +++ b/nexus/model/alibi.py @@ -0,0 +1,137 @@ +""" +ALiBi (Attention with Linear Biases) position bias for Nexus Coder v0.3 +====================================================================== +Alternative to RoPE. No positional embeddings — biases are added directly +to attention scores. Extrapolates better to longer sequences than RoPE. + +Reference: Press et al., "Train Short, Test Long: Attention with Linear +Biases Enables Input Length Extrapolation" (ICLR 2022). +https://arxiv.org/abs/2108.12409 + +Attribution: Algorithm adapted from the original paper. Implementation +references both the original alibi-transformers repo and HuggingFace's +integration in `bloom` / `mntptr` projects. +""" +from __future__ import annotations + +import math +from typing import List + +import torch +import torch.nn as nn + + +def get_alibi_slopes(num_heads: int, max_slope: float = 8.0) -> torch.Tensor: + """Compute ALiBi slopes for `num_heads` attention heads. + + v0.4 fix: use `max_slope` correctly (was hardcoded to 8.0 → log2(8)=3). + v0.4 fix: non-power-of-2 head counts now pick the *closest* n slopes + (standard ALiBi behavior), not "evenly spaced" (which was buggy). + + Args: + num_heads: number of attention heads + max_slope: steepest slope (controls decay). Default 8.0. + + Returns: + slopes: tensor of shape [num_heads] + """ + if num_heads <= 0: + return torch.tensor([], dtype=torch.float32) + + log_max = math.log2(max_slope) # e.g. log2(8)=3 + + def _get_slopes_power_of_2(n: int) -> List[float]: + start = 2.0 ** (-(2.0 ** -(math.log2(n) - log_max))) + return [start * (2.0 ** (-i)) for i in range(n)] + + if (num_heads & (num_heads - 1)) == 0: + # Power of 2 — direct + slopes = _get_slopes_power_of_2(num_heads) + else: + # Non-power-of-2: standard ALiBi picks the n closest slopes + # by computing slopes for the nearest power of 2 >= n and + # interleaving them, then taking the first n. + base = 1 + while base < num_heads: + base *= 2 + full = _get_slopes_power_of_2(base) + # Interleave: take even-indexed first, then odd, to pick "closest" slopes + interleaved = ( + [full[i] for i in range(0, base, 2)] + + [full[i] for i in range(1, base, 2)] + ) + slopes = interleaved[:num_heads] + + return torch.tensor(slopes, dtype=torch.float32) + + +def build_alibi_tensor( + num_heads: int, + seq_len: int, + device: torch.device, + dtype: torch.dtype = torch.float32, + max_slope: float = 8.0, +) -> torch.Tensor: + """Build the additive ALiBi bias tensor. + + Args: + num_heads: number of attention heads + seq_len: attention sequence length + device: target device + dtype: target dtype + max_slope: maximum slope (controls decay) + + Returns: + alibi: tensor of shape [1, num_heads, seq_len, seq_len] + Ready to ADD to attention weights before softmax. + """ + slopes = get_alibi_slopes(num_heads, max_slope=max_slope).to(device=device, dtype=dtype) + # positions: [seq_len, seq_len], value = j - i (j is query, i is key) + positions = torch.arange(seq_len, device=device, dtype=dtype) + relative_positions = positions[None, :] - positions[:, None] # [T, T] + # Mask future positions to -inf (handled by causal mask elsewhere, but be safe) + relative_positions = relative_positions.clamp(min=0) + # alibi: [num_heads, seq_len, seq_len] = -slope * relative_positions + alibi = slopes.view(-1, 1, 1) * relative_positions.unsqueeze(0) + alibi = -alibi # bias is negative (decreases attention with distance) + # Add batch dim + alibi = alibi.unsqueeze(0) # [1, num_heads, seq_len, seq_len] + return alibi.to(dtype=dtype) + + +class AlibiPositionBias(nn.Module): + """Module wrapper for ALiBi bias — registered as buffer, recomputed if seq_len grows.""" + + def __init__(self, num_heads: int, max_slope: float = 8.0): + super().__init__() + self.num_heads = num_heads + self.max_slope = max_slope + slopes = get_alibi_slopes(num_heads, max_slope=max_slope) + self.register_buffer("slopes", slopes, persistent=False) + self._cached_seq_len = 0 + self._cached_bias: torch.Tensor | None = None + + def forward( + self, + seq_len: int, + device: torch.device, + dtype: torch.dtype = torch.float32, + ) -> torch.Tensor: + """Return ALiBi bias of shape [1, num_heads, seq_len, seq_len].""" + if self._cached_bias is None or seq_len > self._cached_seq_len: + self._cached_bias = build_alibi_tensor( + self.num_heads, seq_len, device=device, dtype=dtype, max_slope=self.max_slope, + ) + self._cached_seq_len = seq_len + bias = self._cached_bias.to(device=device, dtype=dtype) + if bias.shape[-1] < seq_len: + # Re-build for new length + self._cached_bias = build_alibi_tensor( + self.num_heads, seq_len, device=device, dtype=dtype, max_slope=self.max_slope, + ) + self._cached_seq_len = seq_len + bias = self._cached_bias + return bias[:, :, :seq_len, :seq_len] + + def extra_repr(self) -> str: + return f"num_heads={self.num_heads}, max_slope={self.max_slope}" diff --git a/nexus/model/attention.py b/nexus/model/attention.py new file mode 100644 index 0000000000000000000000000000000000000000..723c0178fba28676aadfbd563e24da8d901fdfce --- /dev/null +++ b/nexus/model/attention.py @@ -0,0 +1,308 @@ +""" +Multi-Head Attention v0.3 +========================= +Features: + - Grouped Query Attention (GQA) + - RoPE with optional NTK/YaRN scaling (long-context extension) + - FlashAttention-2 backend (when available, falls back to SDPA) + - ALiBi position bias (optional alternative to RoPE) + - Sliding window attention (alternating with global layers) + - QK-norm (RMSNorm on query/key for training stability) + - KV cache quantization (int8/fp8 for memory-efficient inference) + +Author: Hieu Louis (2026) +""" +from __future__ import annotations + +import torch +import torch.nn as nn +import torch.nn.functional as F +from typing import Optional, Tuple + +from .rope import RotaryEmbedding, apply_rotary_pos_emb +from .flash_attention import flash_attention_forward, has_flash_attention_2 +from .alibi import AlibiPositionBias +from .sliding_window import SlidingWindowMaskCache + + +class QKNorm(nn.Module): + """RMSNorm applied to query and key (Llama-3 style).""" + + def __init__(self, head_dim: int, eps: float = 1e-6): + super().__init__() + self.eps = eps + self.weight = nn.Parameter(torch.ones(head_dim)) + + def forward(self, x: torch.Tensor) -> torch.Tensor: + norm = x.float() * torch.rsqrt(x.float().pow(2).mean(-1, keepdim=True) + self.eps) + return (norm.to(x.dtype) * self.weight) + + +class Attention(nn.Module): + """Multi-Head Attention with GQA + RoPE/ALiBi + FlashAttention + sliding window + QK-norm.""" + + def __init__(self, config, layer_idx: int = 0, attention_pattern: str = "global"): + super().__init__() + self.config = config + self.layer_idx = layer_idx + self.attention_pattern = attention_pattern # "global" | "sliding_window" + self.hidden_size = config.hidden_size + self.num_heads = config.num_attention_heads + self.num_kv_heads = config.num_kv_heads + self.head_dim = config.head_dim + self.num_kv_groups = self.num_heads // self.num_kv_heads + + self.q_proj = nn.Linear(self.hidden_size, self.num_heads * self.head_dim, bias=False) + self.k_proj = nn.Linear(self.hidden_size, self.num_kv_heads * self.head_dim, bias=False) + self.v_proj = nn.Linear(self.hidden_size, self.num_kv_heads * self.head_dim, bias=False) + self.o_proj = nn.Linear(self.num_heads * self.head_dim, self.hidden_size, bias=False) + + # === RoPE or ALiBi === + self.use_alibi = config.use_alibi + if not self.use_alibi: + self.rotary_emb = RotaryEmbedding( + dim=self.head_dim, + max_position_embeddings=config.max_position_embeddings, + base=config.rotary_emb_base, + scaling_type=config.rope_scaling_type, + scaling_factor=config.rope_scaling_factor, + yarn_beta_fast=getattr(config, "yarn_beta_fast", 32.0), + yarn_beta_slow=getattr(config, "yarn_beta_slow", 1.0), + ) + else: + self.alibi = AlibiPositionBias( + num_heads=self.num_heads, + max_slope=getattr(config, "alibi_max_slope", 8.0), + ) + + # === QK-norm (Llama-3 style) === + self.use_qk_norm = config.use_qk_norm + if self.use_qk_norm: + self.q_norm = QKNorm(self.head_dim, eps=config.qk_norm_eps) + self.k_norm = QKNorm(self.head_dim, eps=config.qk_norm_eps) + else: + self.q_norm = None + self.k_norm = None + + # === FlashAttention === + self.use_flash_attn_2 = config.use_flash_attention_2 and has_flash_attention_2() + self.use_sdpa = config.use_flash_attention # PyTorch SDPA (always available) + self.attn_dropout = config.attention_dropout + + # === Sliding window mask cache === + self.use_sliding_window = ( + config.use_sliding_window and attention_pattern == "sliding_window" + ) + self.sliding_window_size = config.sliding_window_size + if self.use_sliding_window: + self._swa_cache = SlidingWindowMaskCache(window_size=self.sliding_window_size) + else: + self._swa_cache = None + + # === KV cache quantization === + self.kv_cache_quantization = config.kv_cache_quantization + self.kv_cache_bits = config.kv_cache_bits + + def _quantize_kv_cache(self, x: torch.Tensor): + """Quantize KV cache tensor to int8/fp8 to save memory (only at inference). + + Returns: + - For int8: (quantized_tensor_int8, scale_tensor) + - For fp8: (tensor_fp8, None) + - None / float input: (x, None) + """ + if self.kv_cache_quantization is None or not torch.is_floating_point(x): + return x, None + if self.kv_cache_quantization == "int8": + # Symmetric int8 quantization, scale stored alongside (per-row) + abs_max = x.abs().amax(dim=-1, keepdim=True).clamp(min=1e-8) + scale = abs_max / 127.0 + q = (x / scale).round().clamp(-128, 127).to(torch.int8) + return q, scale + elif self.kv_cache_quantization == "fp8": + return x.to(torch.float8_e4m3fn), None + return x, None + + def _dequantize_kv_cache(self, x, scale=None) -> torch.Tensor: + """Dequantize KV cache back to float (no-op if already float).""" + if self.kv_cache_quantization is None or torch.is_floating_point(x): + return x + if self.kv_cache_quantization == "int8": + if scale is None: + # Cannot recover without scale → return zeros (graceful degradation) + return torch.zeros_like(x, dtype=torch.float32) + return x.to(torch.float32) * scale + elif self.kv_cache_quantization == "fp8": + return x.to(torch.float32) + return x + + def forward( + self, + hidden_states: torch.Tensor, + attention_mask: Optional[torch.Tensor] = None, + position_ids: Optional[torch.Tensor] = None, + past_key_value: Optional[Tuple[torch.Tensor, torch.Tensor]] = None, + use_cache: bool = False, + ) -> Tuple[torch.Tensor, Optional[Tuple[torch.Tensor, torch.Tensor]]]: + bsz, q_len, _ = hidden_states.size() + + query_states = self.q_proj(hidden_states).view( + bsz, q_len, self.num_heads, self.head_dim, + ).transpose(1, 2) + key_states = self.k_proj(hidden_states).view( + bsz, q_len, self.num_kv_heads, self.head_dim, + ).transpose(1, 2) + value_states = self.v_proj(hidden_states).view( + bsz, q_len, self.num_kv_heads, self.head_dim, + ).transpose(1, 2) + + # QK-norm + if self.use_qk_norm: + query_states = self.q_norm(query_states) + key_states = self.k_norm(key_states) + + # Apply RoPE + if not self.use_alibi: + cos, sin = self.rotary_emb(value_states, seq_len=q_len) + query_states, key_states = apply_rotary_pos_emb( + query_states, key_states, cos, sin, position_ids, + ) + + # KV cache + if past_key_value is not None: + # Unpack: past_key_value is (cached_k, cached_v, k_scale, v_scale) for int8 + if isinstance(past_key_value, tuple) and len(past_key_value) == 4: + cached_k, cached_v, k_scale, v_scale = past_key_value + else: + cached_k, cached_v = past_key_value + k_scale, v_scale = None, None + # dequantize if needed + cached_k = self._dequantize_kv_cache(cached_k, k_scale) + cached_v = self._dequantize_kv_cache(cached_v, v_scale) + key_states = torch.cat([cached_k, key_states], dim=2) + value_states = torch.cat([cached_v, value_states], dim=2) + past_key_value = None + if use_cache: + # Quantize for storage (scales preserved) + k_cached, k_scale = self._quantize_kv_cache(key_states) + v_cached, v_scale = self._quantize_kv_cache(value_states) + # Always return 4-tuple so downstream code knows the layout + past_key_value = (k_cached, v_cached, k_scale, v_scale) + + # Repeat K, V cho GQA + if self.num_kv_groups > 1: + key_states = key_states.repeat_interleave(self.num_kv_groups, dim=1) + value_states = value_states.repeat_interleave(self.num_kv_groups, dim=1) + + # Build attention mask + full_mask = None + if self.use_sliding_window and self._swa_cache is not None: + full_seq_len = key_states.shape[2] + full_mask = self._swa_cache.get( + seq_len=full_seq_len, + pattern="sliding_window", + device=hidden_states.device, + dtype=query_states.dtype, + ) + if attention_mask is not None: + # attention_mask: [B, 1, 1, T] (0 = keep, -inf = mask) + full_mask = full_mask + attention_mask + elif attention_mask is not None: + full_mask = attention_mask + + # ALiBi additive bias + if self.use_alibi: + full_seq_len = key_states.shape[2] + alibi_bias = self.alibi( + seq_len=full_seq_len, + device=hidden_states.device, + dtype=query_states.dtype, + ) + # ALiBi is [1, num_heads, T, T]; broadcast + if full_mask is None: + full_mask = alibi_bias + else: + full_mask = full_mask + alibi_bias + + # YaRN temperature correction + softmax_scale = None + if not self.use_alibi and self.config.rope_scaling_type == "yarn": + temperature = self.rotary_emb.get_attention_temperature() + softmax_scale = (self.head_dim ** -0.5) / temperature + + # Compute attention + if self.use_flash_attn_2: + attn_output = flash_attention_forward( + query_states, key_states, value_states, + attention_mask=full_mask, + dropout=self.attn_dropout, + is_causal=True, + use_flash_attn_2=True, + softmax_scale=softmax_scale, + ) + elif self.use_sdpa: + try: + attn_output = F.scaled_dot_product_attention( + query_states, key_states, value_states, + attn_mask=full_mask, + dropout_p=self.attn_dropout if self.training else 0.0, + is_causal=(full_mask is None), + scale=softmax_scale, + ) + except Exception: + # Manual fallback + attn_weights = torch.matmul(query_states, key_states.transpose(2, 3)) + scale = softmax_scale or (self.head_dim ** -0.5) + attn_weights = attn_weights * scale + if full_mask is not None: + attn_weights = attn_weights + full_mask + else: + causal_mask = torch.triu( + torch.full((q_len, q_len), float("-inf"), + device=hidden_states.device, dtype=query_states.dtype), + diagonal=1, + ) + attn_weights = attn_weights + causal_mask + attn_weights = F.softmax(attn_weights, dim=-1, dtype=torch.float32).to(query_states.dtype) + if self.attn_dropout > 0 and self.training: + attn_weights = F.dropout(attn_weights, p=self.attn_dropout) + attn_output = torch.matmul(attn_weights, value_states) + else: + # Manual attention (slow) + attn_weights = torch.matmul(query_states, key_states.transpose(2, 3)) + scale = softmax_scale or (self.head_dim ** -0.5) + attn_weights = attn_weights * scale + if full_mask is not None: + attn_weights = attn_weights + full_mask + else: + causal_mask = torch.triu( + torch.full((q_len, q_len), float("-inf"), + device=hidden_states.device, dtype=query_states.dtype), + diagonal=1, + ) + attn_weights = attn_weights + causal_mask + attn_weights = F.softmax(attn_weights, dim=-1, dtype=torch.float32).to(query_states.dtype) + attn_output = torch.matmul(attn_weights, value_states) + + attn_output = attn_output.transpose(1, 2).contiguous() + attn_output = attn_output.view(bsz, q_len, self.num_heads * self.head_dim) + attn_output = self.o_proj(attn_output) + return attn_output, past_key_value + + def extra_repr(self) -> str: + s = f"heads={self.num_heads} (kv={self.num_kv_heads}), head_dim={self.head_dim}" + if self.use_alibi: + s += ", alibi=ON" + else: + s += f", rope_scaling={self.config.rope_scaling_type or 'none'}" + if self.use_qk_norm: + s += ", qk_norm=ON" + if self.use_flash_attn_2: + s += ", fa2=ON" + elif self.use_sdpa: + s += ", sdpa=ON" + if self.use_sliding_window: + s += f", swa(window={self.sliding_window_size})" + if self.kv_cache_quantization: + s += f", kv_quant={self.kv_cache_quantization}" + return s diff --git a/nexus/model/flash_attention.py b/nexus/model/flash_attention.py new file mode 100644 index 0000000000000000000000000000000000000000..1f1fae72bf518aee45b7960fb49805f6fd554893 --- /dev/null +++ b/nexus/model/flash_attention.py @@ -0,0 +1,154 @@ +""" +FlashAttention-2 wrapper for Nexus Coder v0.3 +============================================= +Provides a unified interface for: + 1. PyTorch native SDPA (F.scaled_dot_product_attention) — always available + 2. FlashAttention-2 (flash_attn package) — optional, faster on Ampere+ + +If `flash_attn` is not installed, we silently fall back to SDPA. + +Attribution: FlashAttention-2 algorithm from Dao et al. (2023). +Reference implementation: https://github.com/Dao-AILab/flash-attention +""" +from __future__ import annotations + +from typing import Optional, Tuple + +import torch +import torch.nn as nn +import torch.nn.functional as F + +try: + # Optional dependency — installed via: pip install flash-attn --no-build-isolation + from flash_attn import flash_attn_func # type: ignore + _HAS_FLASH_ATTN_2 = True +except Exception: + _HAS_FLASH_ATTN_2 = False + + +def has_flash_attention_2() -> bool: + """Check whether the FlashAttention-2 package is available at runtime.""" + return _HAS_FLASH_ATTN_2 + + +def flash_attention_forward( + query_states: torch.Tensor, + key_states: torch.Tensor, + value_states: torch.Tensor, + attention_mask: Optional[torch.Tensor] = None, + dropout: float = 0.0, + is_causal: bool = True, + use_flash_attn_2: bool = False, + softmax_scale: Optional[float] = None, +) -> torch.Tensor: + """Unified entry point for attention computation. + + Args: + query_states: [B, num_heads, T, head_dim] (SDPA layout) + or [B, T, num_heads, head_dim] (FA2 layout, if use_flash_attn_2) + key_states: same layout as query_states + value_states: same layout as query_states + attention_mask: optional additive mask (SDPA only). Ignored for FA2. + dropout: attention dropout probability + is_causal: whether to apply causal mask + use_flash_attn_2: try to use FlashAttention-2 (falls back to SDPA if unavailable) + softmax_scale: custom scale; default = head_dim ** -0.5 + + Returns: + attn_output: same layout as input + """ + head_dim = query_states.shape[-1] + if softmax_scale is None: + softmax_scale = head_dim ** -0.5 + + # === FlashAttention-2 path === + if use_flash_attn_2 and _HAS_FLASH_ATTN_2 and not attention_mask is not None: + # FA2 expects [B, T, num_heads, head_dim] + if query_states.dim() == 4 and query_states.shape[1] != query_states.shape[2]: + # Likely [B, num_heads, T, head_dim] — transpose + q = query_states.transpose(1, 2) + k = key_states.transpose(1, 2) + v = value_states.transpose(1, 2) + else: + q, k, v = query_states, key_states, value_states + out = flash_attn_func( + q, k, v, + dropout_p=dropout if torch.is_grad_enabled() else 0.0, + softmax_scale=softmax_scale, + causal=is_causal, + ) + # Convert back to [B, num_heads, T, head_dim] + if out.shape[1] != query_states.shape[1] if query_states.dim() == 4 else True: + out = out.transpose(1, 2) + return out + + # === PyTorch SDPA path (always available) === + # SDPA supports attn_mask as additive bias + try: + out = F.scaled_dot_product_attention( + query_states, + key_states, + value_states, + attn_mask=attention_mask, + dropout_p=dropout if torch.is_grad_enabled() else 0.0, + is_causal=is_causal and attention_mask is None, + scale=softmax_scale, + ) + return out + except Exception: + # Manual fallback (very slow, for debugging only) + attn_weights = torch.matmul(query_states, key_states.transpose(-2, -1)) * softmax_scale + if is_causal and attention_mask is None: + T = attn_weights.shape[-2] + causal_mask = torch.triu( + torch.full((T, T), float("-inf"), device=attn_weights.device, dtype=attn_weights.dtype), + diagonal=1, + ) + attn_weights = attn_weights + causal_mask + elif attention_mask is not None: + attn_weights = attn_weights + attention_mask + attn_weights = F.softmax(attn_weights, dim=-1, dtype=torch.float32).to(query_states.dtype) + if dropout > 0 and torch.is_grad_enabled(): + attn_weights = F.dropout(attn_weights, p=dropout) + return torch.matmul(attn_weights, value_states) + + +class FlashAttention(nn.Module): + """Drop-in replacement for the manual attention in `nexus/model/attention.py`. + + Automatically picks the best available backend: + - FlashAttention-2 if `use_flash_attn_2=True` and package is installed + - F.scaled_dot_product_attention (SDPA) otherwise + - Manual fallback as last resort + """ + + def __init__( + self, + use_flash_attn_2: bool = False, + dropout: float = 0.0, + softmax_scale: Optional[float] = None, + ): + super().__init__() + self.use_flash_attn_2 = use_flash_attn_2 and _HAS_FLASH_ATTN_2 + self.dropout = dropout + self.softmax_scale = softmax_scale + + def forward( + self, + q: torch.Tensor, + k: torch.Tensor, + v: torch.Tensor, + attention_mask: Optional[torch.Tensor] = None, + is_causal: bool = True, + ) -> torch.Tensor: + return flash_attention_forward( + q, k, v, + attention_mask=attention_mask, + dropout=self.dropout, + is_causal=is_causal, + use_flash_attn_2=self.use_flash_attn_2, + softmax_scale=self.softmax_scale, + ) + + def extra_repr(self) -> str: + return f"flash_attn_2={self.use_flash_attn_2}, dropout={self.dropout}" diff --git a/nexus/model/layers.py b/nexus/model/layers.py new file mode 100644 index 0000000000000000000000000000000000000000..2cd9d1dce6dfe6b2c63876b6c531271089853608 --- /dev/null +++ b/nexus/model/layers.py @@ -0,0 +1,72 @@ +""" +RMSNorm + SwiGLU layers v0.3 +============================ +- RMSNorm (Zhang & Sennrich, 2019) — unchanged +- SwiGLU — adds MLP-parallel variant (compute gate/up in parallel) +""" +from __future__ import annotations + +import torch +import torch.nn as nn +import torch.nn.functional as F + + +class RMSNorm(nn.Module): + """Root Mean Square LayerNorm (Zhang & Sennrich, 2019). + Hiệu quả hơn LayerNorm truyền thống, không có bias và không trừ mean. + """ + + def __init__(self, hidden_size: int, eps: float = 1e-6): + super().__init__() + self.weight = nn.Parameter(torch.ones(hidden_size)) + self.eps = eps + + def forward(self, hidden_states: torch.Tensor) -> torch.Tensor: + input_dtype = hidden_states.dtype + hidden_states = hidden_states.to(torch.float32) + variance = hidden_states.pow(2).mean(-1, keepdim=True) + hidden_states = hidden_states * torch.rsqrt(variance + self.eps) + return self.weight * hidden_states.to(input_dtype) + + +class SwiGLU(nn.Module): + """SwiGLU activation: SiLU(gate(x)) * up(x). + + v0.3: adds MLP-parallel variant — gate_proj and up_proj are computed + as a single concatenated matmul (faster on modern GPUs). + """ + + def __init__(self, hidden_size: int, intermediate_size: int, parallel: bool = True): + super().__init__() + self.parallel = parallel + if parallel: + # Concatenated gate + up projection (mathematically identical, faster) + self.gate_up_proj = nn.Linear( + hidden_size, 2 * intermediate_size, bias=False, + ) + self.gate_proj = None + self.up_proj = None + else: + self.gate_proj = nn.Linear(hidden_size, intermediate_size, bias=False) + self.up_proj = nn.Linear(hidden_size, intermediate_size, bias=False) + self.gate_up_proj = None + self.down_proj = nn.Linear(intermediate_size, hidden_size, bias=False) + self.intermediate_size = intermediate_size + + def forward(self, x: torch.Tensor) -> torch.Tensor: + if self.parallel: + gate_up = self.gate_up_proj(x) + gate, up = gate_up[..., : self.intermediate_size], gate_up[..., self.intermediate_size :] + gate = F.silu(gate) + else: + gate = F.silu(self.gate_proj(x)) + up = self.up_proj(x) + return self.down_proj(gate * up) + + +def _expand_token_ids_to_mask(token_ids: torch.Tensor, seq_len: int) -> torch.Tensor: + """Helper: chuyển token ids thành attention mask.""" + mask = torch.zeros(token_ids.shape[0], seq_len, device=token_ids.device) + for i, ids in enumerate(token_ids): + mask[i, : len(ids)] = 1 + return mask diff --git a/nexus/model/moe.py b/nexus/model/moe.py new file mode 100644 index 0000000000000000000000000000000000000000..fb6cfc3d9bc91191396c001ed5b4455f9635e879 --- /dev/null +++ b/nexus/model/moe.py @@ -0,0 +1,189 @@ +""" +Mixture of Experts (MoE) Layer - Cốt lõi của Nexus Coder +========================================================= +24 chuyên gia (experts) tổng cộng, chỉ 3 chuyên gia được kích hoạt mỗi token. +Đạt được 10B tổng tham số với chỉ 1.5B tham số active. + +Tính năng: +- Top-K routing với noise (load balancing) +- Aux loss cho load balancing giữa các expert +- Hỗ trợ SwiGLU experts +""" +import torch +import torch.nn as nn +import torch.nn.functional as F +from typing import Tuple, Optional + +from .layers import SwiGLU + + +class Expert(nn.Module): + """Một chuyên gia (expert) - thực chất là một SwiGLU FFN. + + v0.3: hỗ trợ MLP-parallel (gate/up concat thành 1 matmul). + """ + + def __init__(self, hidden_size: int, intermediate_size: int, parallel: bool = True): + super().__init__() + self.ffn = SwiGLU(hidden_size, intermediate_size, parallel=parallel) + + def forward(self, x: torch.Tensor) -> torch.Tensor: + return self.ffn(x) + + +class Router(nn.Module): + """Router/Gating network: quyết định token nào đi đến expert nào.""" + + def __init__(self, hidden_size: int, num_experts: int): + super().__init__() + self.gate = nn.Linear(hidden_size, num_experts, bias=False) + + def forward(self, x: torch.Tensor) -> torch.Tensor: + return self.gate(x) + + +def load_balancing_loss_func( + gate_logits: torch.Tensor, + num_experts: int, + top_k: int, + attention_mask: Optional[torch.Tensor] = None, +) -> torch.Tensor: + """Tính auxiliary loss cho load balancing (Switch Transformer). + + attention_mask có thể là: + - None: tất cả token đều valid + - 2D bool [B, T]: True = valid token + - 2D int [B, T]: 1 = valid, 0 = padding + - 4D float [B, 1, 1, T]: 0 = valid, large_negative = padding + """ + if gate_logits is None: + # gate_logits is None → cannot compute; return 0 on proper device + return torch.tensor(0.0) + + # Normalize attention_mask → 1D bool [N_valid] + if attention_mask is None: + tokens_per_expert = gate_logits.shape[0] * gate_logits.shape[1] + # 2D shape: [B, T] already flattened by caller, so gate_logits.shape[0] is N + if gate_logits.dim() == 2: + tokens_per_expert = gate_logits.shape[0] + else: + # Convert 4D mask to 2D bool + if attention_mask.dim() == 4: + # [B, 1, 1, T] with 0 / -inf values + mask_2d = attention_mask.squeeze(1).squeeze(1) # [B, T] + mask_bool = mask_2d > -1e9 + elif attention_mask.dim() == 3: + mask_bool = attention_mask.squeeze(1) > 0 + elif attention_mask.dim() == 2: + if attention_mask.dtype == torch.bool: + mask_bool = attention_mask + else: + # 0/1 or 0/-inf + if attention_mask.dtype.is_floating_point: + mask_bool = attention_mask > -1e9 + else: + mask_bool = attention_mask > 0 + else: + mask_bool = None + + if mask_bool is None: + tokens_per_expert = gate_logits.shape[0] + else: + tokens_per_expert = mask_bool.sum().item() + if tokens_per_expert < 1: + tokens_per_expert = gate_logits.shape[0] + + routing_weights = F.softmax(gate_logits, dim=-1) + _, selected_experts = torch.topk(routing_weights, top_k, dim=-1) + + expert_mask = F.one_hot(selected_experts, num_classes=num_experts) + expert_mask = expert_mask.sum(dim=-2).float() + + tokens_per_expert_normalized = expert_mask.mean(dim=-2) + router_prob_per_expert = routing_weights.mean(dim=-2) + + aux_loss = ( + num_experts * (tokens_per_expert_normalized * router_prob_per_expert).sum() + ) / max(tokens_per_expert, 1) + + return aux_loss + + +class MixtureOfExperts(nn.Module): + """MoE Layer với Top-K routing và load balancing.""" + + def __init__(self, config): + super().__init__() + self.config = config + self.num_experts = config.num_experts + self.num_active_experts = config.num_active_experts + self.router_jitter_noise = config.router_jitter_noise + self.aux_loss_coef = config.router_aux_loss_coef + + # Router + self.router = Router(config.hidden_size, self.num_experts) + + # Experts (v0.3: MLP-parallel by default) + mlp_parallel = getattr(config, "mlp_parallel", True) + self.experts = nn.ModuleList([ + Expert(config.hidden_size, config.intermediate_size, parallel=mlp_parallel) + for _ in range(self.num_experts) + ]) + + def forward( + self, + hidden_states: torch.Tensor, + attention_mask: Optional[torch.Tensor] = None, + ) -> Tuple[torch.Tensor, torch.Tensor]: + bsz, seq_len, hidden = hidden_states.shape + flat_hidden = hidden_states.view(-1, hidden) # [N, H] + + # Router logits + router_logits = self.router(flat_hidden) # [N, E] + + # Thêm noise trong training để encourage exploration + if self.training and self.router_jitter_noise > 0: + router_logits = router_logits + torch.randn_like(router_logits) * self.router_jitter_noise + + # Top-K routing + routing_weights = F.softmax(router_logits, dim=-1) + top_k_weights, top_k_indices = torch.topk( + routing_weights, self.num_active_experts, dim=-1 + ) + top_k_weights = top_k_weights / (top_k_weights.sum(dim=-1, keepdim=True) + 1e-9) + + # Dispatch tokens to experts + final_hidden = torch.zeros_like(flat_hidden) + + # Vectorized: iterate through experts + for expert_idx in range(self.num_experts): + # Find tokens that go to this expert + expert_mask = (top_k_indices == expert_idx).any(dim=-1) # [N] + if not expert_mask.any(): + continue + + # Get token indices + token_indices = expert_mask.nonzero(as_tuple=True)[0] + + # Get the corresponding weights + expert_weights = top_k_weights[token_indices] # [num_tokens, top_k] + expert_weight_for_this = (top_k_indices[token_indices] == expert_idx).float() * expert_weights + expert_weight_for_this = expert_weight_for_this.sum(dim=-1) # [num_tokens] + + # Run expert + expert_input = flat_hidden[token_indices] + expert_output = self.experts[expert_idx](expert_input) + expert_output = expert_output * expert_weight_for_this.unsqueeze(-1) + + final_hidden[token_indices] += expert_output + + # Load balancing loss + aux_loss = load_balancing_loss_func( + router_logits, + self.num_experts, + self.num_active_experts, + attention_mask, + ) + + final_hidden = final_hidden.view(bsz, seq_len, hidden) + return final_hidden, aux_loss diff --git a/nexus/model/nexus_coder.py b/nexus/model/nexus_coder.py new file mode 100644 index 0000000000000000000000000000000000000000..bb08db175d764878ce32036993b320ef3b30aeb2 --- /dev/null +++ b/nexus/model/nexus_coder.py @@ -0,0 +1,255 @@ +""" +Nexus Coder Model - Model AI MoE chính +======================================== +Model: Nexus Coder v0.1 +Tác giả: Hieu Louis (2026) + +Đặc điểm: +- 10 tỷ tham số tổng (10B total) +- 1.5 tỷ tham số kích hoạt (1.5B active per token) +- Context window: 50,000 tokens +- Kiến trúc: MoE Transformer với 24 experts, 3 active +- RoPE position embedding +- RMSNorm (pre-norm) +- SwiGLU activation +- GQA (Grouped Query Attention) +""" +import math +import torch +import torch.nn as nn +import torch.nn.functional as F +from typing import Optional, Tuple, Dict, List, Union + +from ..config import NexusConfig +from .layers import RMSNorm +from .transformer import NexusDecoderLayer +from .moe import load_balancing_loss_func +from .sliding_window import get_layer_attention_pattern + + +class NexusCoder(nn.Module): + """Base Nexus Coder model - trả về hidden states.""" + + def __init__(self, config: NexusConfig): + super().__init__() + self.config = config + + # Token embeddings + self.embed_tokens = nn.Embedding(config.vocab_size, config.hidden_size) + + # Per-layer attention pattern: alternating SWA / global + layer_patterns = get_layer_attention_pattern( + num_layers=config.num_hidden_layers, + use_sliding_window=config.use_sliding_window, + sliding_window_layers=config.sliding_window_layers, + ) + + # Decoder layers + self.layers = nn.ModuleList([ + NexusDecoderLayer( + config, + layer_idx=i, + attention_pattern=layer_patterns[i], + ) + for i in range(config.num_hidden_layers) + ]) + + # Final norm + self.norm = RMSNorm(config.hidden_size, eps=config.layer_norm_epsilon) + + def enable_gradient_checkpointing(self): + """Enable gradient checkpointing on all layers.""" + for layer in self.layers: + layer.gradient_checkpointing = True + + def disable_gradient_checkpointing(self): + """Disable gradient checkpointing on all layers.""" + for layer in self.layers: + layer.gradient_checkpointing = False + + def forward( + self, + input_ids: torch.Tensor, + attention_mask: Optional[torch.Tensor] = None, + position_ids: Optional[torch.Tensor] = None, + past_key_values: Optional[List[Tuple[torch.Tensor, torch.Tensor]]] = None, + use_cache: bool = False, + ) -> Tuple[torch.Tensor, Dict]: + bsz, seq_len = input_ids.shape + + if position_ids is None: + position_ids = torch.arange(seq_len, device=input_ids.device).unsqueeze(0).expand(bsz, -1) + + # Embedding + hidden_states = self.embed_tokens(input_ids) + + # Prepare attention mask (causal) + if attention_mask is None: + # Default causal mask + attn_mask = torch.triu( + torch.full((seq_len, seq_len), float("-inf"), device=hidden_states.device), + diagonal=1, + ) + attn_mask = attn_mask.unsqueeze(0).unsqueeze(0) + else: + attn_mask = self._prepare_attention_mask(attention_mask, seq_len) + + # Through layers + all_aux_loss = torch.tensor(0.0, device=hidden_states.device) + new_kv_list = [] + for i, layer in enumerate(self.layers): + past_kv = past_key_values[i] if past_key_values is not None else None + hidden_states, new_kv, aux_loss = layer( + hidden_states, + attention_mask=attn_mask, + position_ids=position_ids, + past_key_value=past_kv, + use_cache=use_cache, + ) + all_aux_loss = all_aux_loss + aux_loss + new_kv_list.append(new_kv) + + # Final norm + hidden_states = self.norm(hidden_states) + + outputs = { + "last_hidden_state": hidden_states, + "aux_loss": all_aux_loss / len(self.layers), + "past_key_values": new_kv_list if use_cache else None, + } + return hidden_states, outputs + + def _prepare_attention_mask(self, attention_mask: torch.Tensor, seq_len: int) -> torch.Tensor: + """Tạo attention mask 4D từ mask 2D.""" + # attention_mask: [B, seq_len] (1 = valid, 0 = padding) + extended = attention_mask[:, None, None, :] + extended = extended.to(dtype=torch.float32) + extended = (1.0 - extended) * torch.finfo(torch.float32).min + return extended + + +class NexusCoderForCausalLM(nn.Module): + """Nexus Coder cho causal language modeling (next-token prediction).""" + + def __init__(self, config: NexusConfig): + super().__init__() + self.config = config + self.model = NexusCoder(config) + + # LM head (không tie weights) + self.lm_head = nn.Linear(config.hidden_size, config.vocab_size, bias=False) + + def forward( + self, + input_ids: torch.Tensor, + attention_mask: Optional[torch.Tensor] = None, + position_ids: Optional[torch.Tensor] = None, + past_key_values: Optional[List[Tuple[torch.Tensor, torch.Tensor]]] = None, + labels: Optional[torch.Tensor] = None, + use_cache: bool = False, + ) -> Dict[str, torch.Tensor]: + hidden_states, outputs = self.model( + input_ids=input_ids, + attention_mask=attention_mask, + position_ids=position_ids, + past_key_values=past_key_values, + use_cache=use_cache, + ) + + # LM head + logits = self.lm_head(hidden_states) + + loss = None + if labels is not None: + # Shift for next token prediction + shift_logits = logits[..., :-1, :].contiguous() + shift_labels = labels[..., 1:].contiguous() + + loss_fct = nn.CrossEntropyLoss() + loss = loss_fct( + shift_logits.view(-1, self.config.vocab_size), + shift_labels.view(-1), + ) + + # Add aux loss + loss = loss + self.config.router_aux_loss_coef * outputs["aux_loss"] + + return { + "loss": loss, + "logits": logits, + "aux_loss": outputs["aux_loss"], + "past_key_values": outputs["past_key_values"], + } + + @torch.no_grad() + def generate( + self, + input_ids: torch.Tensor, + max_new_tokens: int = 100, + temperature: float = 0.8, + top_k: int = 50, + top_p: float = 0.9, + do_sample: bool = True, + pad_token_id: int = 0, + eos_token_id: int = 2, + ) -> torch.Tensor: + """Hàm generate đơn giản với top-k và top-p sampling.""" + self.eval() + device = input_ids.device + + for _ in range(max_new_tokens): + # Forward pass + outputs = self.forward( + input_ids=input_ids, + use_cache=False, + ) + logits = outputs["logits"] + next_logits = logits[:, -1, :] / max(temperature, 1e-8) + + # Top-k + if top_k > 0: + top_k = min(top_k, next_logits.size(-1)) + values, _ = torch.topk(next_logits, top_k) + min_values = values[:, -1].unsqueeze(-1) + next_logits = torch.where( + next_logits < min_values, + torch.full_like(next_logits, float("-inf")), + next_logits, + ) + + # Top-p + if 0 < top_p < 1.0: + sorted_logits, sorted_indices = torch.sort(next_logits, descending=True) + cum_probs = F.softmax(sorted_logits, dim=-1).cumsum(dim=-1) + sorted_indices_to_remove = cum_probs > top_p + sorted_indices_to_remove[..., 1:] = sorted_indices_to_remove[..., :-1].clone() + sorted_indices_to_remove[..., 0] = False + indices_to_remove = sorted_indices_to_remove.scatter( + 1, sorted_indices, sorted_indices_to_remove + ) + next_logits = next_logits.masked_fill(indices_to_remove, float("-inf")) + + # Sample + if do_sample: + probs = F.softmax(next_logits, dim=-1) + next_token = torch.multinomial(probs, num_samples=1) + else: + next_token = torch.argmax(next_logits, dim=-1, keepdim=True) + + input_ids = torch.cat([input_ids, next_token], dim=-1) + + if next_token.item() == eos_token_id: + break + + return input_ids + + def count_parameters(self) -> dict: + """Đếm tham số.""" + total = sum(p.numel() for p in self.parameters()) + trainable = sum(p.numel() for p in self.parameters() if p.requires_grad) + return { + "total": total, + "trainable": trainable, + "total_billion": total / 1e9, + "trainable_billion": trainable / 1e9, + } diff --git a/nexus/model/rope.py b/nexus/model/rope.py new file mode 100644 index 0000000000000000000000000000000000000000..ce8becda519a08ec6eef37445ca12191b6d32bb3 --- /dev/null +++ b/nexus/model/rope.py @@ -0,0 +1,197 @@ +""" +Rotary Position Embedding (RoPE) v0.3 — with NTK-aware + YaRN scaling +==================================================================== +v0.1: basic RoPE (Su et al., 2021) +v0.2: cached cos/sin, max 50k context +v0.3: adds 4 RoPE scaling strategies for context extension: + - "linear": naive linear interpolation (Chen et al., 2023) + - "dynamic": NTK-aware (PureDynamicNTKScaling) — better for short→long + - "ntk": NTK-by-parts (bloc97, 2023) + - "yarn": YaRN (Peng et al., 2023) — SOTA for 4×+ extension + +References: + - Original RoPE: https://arxiv.org/abs/2104.09864 + - YaRN: https://arxiv.org/abs/2309.00071 + - NTK-aware: https://www.reddit.com/r/LocalLLaMA/comments/14lzrgj/ +""" +from __future__ import annotations + +import math +from typing import Optional, Tuple + +import torch +import torch.nn as nn + + +# ============================================================================= +# Scaling strategies +# ============================================================================= + +def _linear_inv_freq(base: float, dim: int, scaling_factor: float) -> torch.Tensor: + """Linear scaling: compress positions by `scaling_factor`.""" + inv_freq = 1.0 / (base ** (torch.arange(0, dim, 2).float() / dim)) + return inv_freq / scaling_factor + + +def _ntk_aware_inv_freq(base: float, dim: int, scaling_factor: float) -> torch.Tensor: + """NTK-aware scaling — modifies base frequency directly. + Better preserves high-frequency components than linear. + """ + base = base * (scaling_factor ** (dim / (dim - 2))) + return 1.0 / (base ** (torch.arange(0, dim, 2).float() / dim)) + + +def _yarn_inv_freq( + base: float, + dim: int, + scaling_factor: float, + beta_fast: float = 32.0, + beta_slow: float = 1.0, +) -> torch.Tensor: + """YaRN scaling — interpolated NTK with attention-factor correction. + Currently we only return the modified inv_freq; the attention factor + correction (temperature) is applied separately in the Attention module. + """ + # Find wavelength boundaries + def _find_correction_dim(num_rot: int, dim: int, base: float, max_seq_len: int) -> float: + return (dim * math.log(max_seq_len / (num_rot * 2 * math.pi))) / (2 * math.log(base)) + + def _find_correction_range( + low_rot: float, high_rot: float, dim: int, base: float, max_seq_len: int, + ) -> Tuple[int, int]: + low = max(math.floor(_find_correction_dim(low_rot, dim, base, max_seq_len)), 0) + high = min(math.ceil(_find_correction_dim(high_rot, dim, base, max_seq_len)), dim - 1) + return low, high + + def _linear_ramp_mask(min_val: float, max_val: float, dim: int) -> torch.Tensor: + if min_val == max_val: + return torch.ones(dim) if min_val > 0 else torch.zeros(dim) + lin = torch.linspace(0, 1, dim) + return torch.clamp((lin - min_val) / (max_val - min_val), 0.0, 1.0) + + max_seq_len = int(4096 * scaling_factor) + low, high = _find_correction_range(beta_fast, beta_slow, dim, base, max_seq_len) + inv_freq_extrapolation = 1.0 / (base ** (torch.arange(0, dim, 2).float() / dim)) + inv_freq_interpolation = 1.0 / (scaling_factor * base ** (torch.arange(0, dim, 2).float() / dim)) + mask = _linear_ramp_mask(low, high, dim // 2).float() + inv_freq = inv_freq_interpolation * mask + inv_freq_extrapolation * (1 - mask) + return inv_freq + + +def compute_inv_freq_with_scaling( + base: float, + dim: int, + scaling_type: Optional[str], + scaling_factor: float, + yarn_beta_fast: float = 32.0, + yarn_beta_slow: float = 1.0, +) -> torch.Tensor: + """Compute inv_freq with the requested scaling strategy.""" + if scaling_type is None or scaling_factor == 1.0: + return 1.0 / (base ** (torch.arange(0, dim, 2).float() / dim)) + if scaling_type == "linear": + return _linear_inv_freq(base, dim, scaling_factor) + if scaling_type == "dynamic": + return _ntk_aware_inv_freq(base, dim, scaling_factor) + if scaling_type == "ntk": + return _ntk_aware_inv_freq(base, dim, scaling_factor) + if scaling_type == "yarn": + return _yarn_inv_freq( + base, dim, scaling_factor, + beta_fast=yarn_beta_fast, beta_slow=yarn_beta_slow, + ) + raise ValueError(f"Unknown rope_scaling_type: {scaling_type}") + + +# ============================================================================= +# Rotary embedding module +# ============================================================================= + +class RotaryEmbedding(nn.Module): + """Rotary Position Embedding with optional scaling (v0.3).""" + + def __init__( + self, + dim: int, + max_position_embeddings: int = 50000, + base: float = 10000.0, + scaling_type: Optional[str] = None, + scaling_factor: float = 1.0, + yarn_beta_fast: float = 32.0, + yarn_beta_slow: float = 1.0, + device: Optional[torch.device] = None, + ): + super().__init__() + self.dim = dim + self.max_position_embeddings = max_position_embeddings + self.base = base + self.scaling_type = scaling_type + self.scaling_factor = scaling_factor + self.yarn_beta_fast = yarn_beta_fast + self.yarn_beta_slow = yarn_beta_slow + + inv_freq = compute_inv_freq_with_scaling( + base=base, + dim=dim, + scaling_type=scaling_type, + scaling_factor=scaling_factor, + yarn_beta_fast=yarn_beta_fast, + yarn_beta_slow=yarn_beta_slow, + ) + self.register_buffer("inv_freq", inv_freq, persistent=False) + self._set_cos_sin_cache( + seq_len=max_position_embeddings, device=device, dtype=torch.get_default_dtype(), + ) + + def _set_cos_sin_cache(self, seq_len: int, device: Optional[torch.device], dtype: torch.dtype): + self.max_seq_len_cached = seq_len + t = torch.arange(seq_len, device=device, dtype=torch.float32) + freqs = torch.einsum("i,j->ij", t, self.inv_freq) + emb = torch.cat([freqs, freqs], dim=-1) + self.register_buffer("cos_cached", emb.cos().to(dtype), persistent=False) + self.register_buffer("sin_cached", emb.sin().to(dtype), persistent=False) + + def forward(self, x: torch.Tensor, seq_len: Optional[int] = None): + if seq_len is None: + seq_len = x.shape[-2] + if seq_len > self.max_seq_len_cached: + self._set_cos_sin_cache(seq_len=seq_len, device=x.device, dtype=x.dtype) + return ( + self.cos_cached[:seq_len, ...].to(x.dtype), + self.sin_cached[:seq_len, ...].to(x.dtype), + ) + + def get_attention_temperature(self) -> float: + """YaRN requires a temperature correction on the attention scores. + Returns the multiplier (1.0 for non-YaRN).""" + if self.scaling_type == "yarn": + # Standard YaRN correction: 0.1 * log(scaling_factor) + 1 + return 0.1 * math.log(self.scaling_factor) + 1.0 + return 1.0 + + +def rotate_half(x: torch.Tensor) -> torch.Tensor: + """Xoay một nửa tensor.""" + x1 = x[..., : x.shape[-1] // 2] + x2 = x[..., x.shape[-1] // 2 :] + return torch.cat((-x2, x1), dim=-1) + + +def apply_rotary_pos_emb( + q: torch.Tensor, + k: torch.Tensor, + cos: torch.Tensor, + sin: torch.Tensor, + position_ids: Optional[torch.Tensor] = None, +) -> Tuple[torch.Tensor, torch.Tensor]: + """Áp dụng RoPE cho q và k.""" + if position_ids is not None: + cos = cos[position_ids].unsqueeze(1) + sin = sin[position_ids].unsqueeze(1) + else: + cos = cos.unsqueeze(0).unsqueeze(0) + sin = sin.unsqueeze(0).unsqueeze(0) + + q_embed = (q * cos) + (rotate_half(q) * sin) + k_embed = (k * cos) + (rotate_half(k) * sin) + return q_embed, k_embed diff --git a/nexus/model/sliding_window.py b/nexus/model/sliding_window.py new file mode 100644 index 0000000000000000000000000000000000000000..c4bff88eb6326958cf23684bbd81dcccbbf8c665 --- /dev/null +++ b/nexus/model/sliding_window.py @@ -0,0 +1,142 @@ +""" +Sliding Window Attention for Nexus Coder v0.3 +============================================= +Local attention within a window of `sliding_window_size` tokens. +Combined with global attention layers, this enables efficient long-context +training (e.g. 64k+ sequences) at a fraction of the compute cost. + +Reference: Beltagy et al., "Longformer: The Long-Document Transformer" (2020). +Attribution: Concept from Longformer / Mistral-7B / Gemma. + +This module exports a helper that builds the appropriate attention mask: + - For SWA layers: causal + windowed (tokens outside the window are masked to -inf) + - For global layers: causal only +""" +from __future__ import annotations + +from typing import List, Optional + +import torch + + +def build_sliding_window_mask( + seq_len: int, + window_size: int, + device: torch.device, + dtype: torch.dtype = torch.float32, + is_causal: bool = True, +) -> torch.Tensor: + """Build a [seq_len, seq_len] additive mask for sliding-window attention. + + A token at position `i` can attend to positions `[max(0, i - window + 1), i]` + (if causal) or `[i - window + 1, i + window - 1]` (non-causal). + + Returns: + mask: tensor of shape [seq_len, seq_len], 0 where allowed and -inf where masked. + """ + # Default: allow everything, then mask out + mask = torch.zeros(seq_len, seq_len, device=device, dtype=dtype) + + if is_causal: + # Causal: can only look at past + self + causal_mask = torch.triu( + torch.full((seq_len, seq_len), float("-inf"), device=device, dtype=dtype), + diagonal=1, + ) + mask = mask + causal_mask + + # Sliding window: mask positions outside [i - window + 1, i] (causal) or + # [i - window + 1, i + window - 1] (non-causal) + for i in range(seq_len): + if is_causal: + lo = max(0, i - window_size + 1) + hi = i + 1 + # Mask everything outside [lo, hi] + if lo > 0: + mask[i, :lo] = float("-inf") + else: + lo = max(0, i - window_size + 1) + hi = min(seq_len, i + window_size) + if lo > 0: + mask[i, :lo] = float("-inf") + if hi < seq_len: + mask[i, hi:] = float("-inf") + + return mask + + +def get_layer_attention_pattern( + num_layers: int, + use_sliding_window: bool, + sliding_window_layers: Optional[List[int]] = None, +) -> List[str]: + """Decide which layers use SWA vs global attention. + + Mistral-7B alternates: SWA on even layers, global on odd. + We follow the same convention if `sliding_window_layers` is None. + + Returns: + List of strings: "sliding_window" or "global", one per layer. + """ + if not use_sliding_window: + return ["global"] * num_layers + if sliding_window_layers is not None: + return [ + "sliding_window" if i in sliding_window_layers else "global" + for i in range(num_layers) + ] + # Default: alternate SWA / global + return [ + "sliding_window" if i % 2 == 0 else "global" + for i in range(num_layers) + ] + + +def apply_pattern_to_mask( + seq_len: int, + window_size: int, + pattern: str, + device: torch.device, + dtype: torch.dtype = torch.float32, +) -> torch.Tensor: + """Build the mask for a single layer based on its pattern.""" + if pattern == "sliding_window": + return build_sliding_window_mask( + seq_len=seq_len, + window_size=window_size, + device=device, + dtype=dtype, + is_causal=True, + ) + # global: causal only + causal = torch.triu( + torch.full((seq_len, seq_len), float("-inf"), device=device, dtype=dtype), + diagonal=1, + ) + return causal + + +class SlidingWindowMaskCache: + """Caches sliding-window masks per layer pattern to avoid recompute.""" + + def __init__(self, window_size: int): + self.window_size = window_size + self._cache: dict[tuple[int, str, torch.device, torch.dtype], torch.Tensor] = {} + + def get( + self, + seq_len: int, + pattern: str, + device: torch.device, + dtype: torch.dtype = torch.float32, + ) -> torch.Tensor: + key = (seq_len, pattern, device, dtype) + if key not in self._cache: + self._cache[key] = apply_pattern_to_mask( + seq_len=seq_len, + window_size=self.window_size, + pattern=pattern, + device=device, + dtype=dtype, + ) + return self._cache[key] diff --git a/nexus/model/transformer.py b/nexus/model/transformer.py new file mode 100644 index 0000000000000000000000000000000000000000..50c57494760cad0bc54187e194682809f54bdd1a --- /dev/null +++ b/nexus/model/transformer.py @@ -0,0 +1,114 @@ +""" +Transformer Decoder Block v0.3 +============================== +Kết hợp Attention (with SWA pattern) + MoE + RMSNorm với pre-norm structure. + +v0.3 NEW: +- Per-layer attention pattern (sliding_window vs global) +- Gradient checkpointing hook (saves VRAM on long context) +- MoE layer accepts MLP-parallel experts +""" +from __future__ import annotations + +import torch +import torch.nn as nn +from typing import Optional, Tuple + +from .layers import RMSNorm +from .attention import Attention +from .moe import MixtureOfExperts +from .sliding_window import get_layer_attention_pattern + + +class NexusDecoderLayer(nn.Module): + """Một decoder layer với: Attention → MoE, cả hai có residual + pre-norm.""" + + def __init__(self, config, layer_idx: int = 0, attention_pattern: str = "global"): + super().__init__() + self.layer_idx = layer_idx + self.config = config + self.attention_pattern = attention_pattern + + # Pre-norm + self.input_norm = RMSNorm(config.hidden_size, eps=config.layer_norm_epsilon) + self.post_attention_norm = RMSNorm(config.hidden_size, eps=config.layer_norm_epsilon) + + # Attention with layer pattern + self.self_attn = Attention( + config, layer_idx=layer_idx, attention_pattern=attention_pattern, + ) + + # MoE FFN + self.moe = MixtureOfExperts(config) + + # Gradient checkpointing flag (set on the parent model) + self.gradient_checkpointing = False + + def forward( + self, + hidden_states: torch.Tensor, + attention_mask: Optional[torch.Tensor] = None, + position_ids: Optional[torch.Tensor] = None, + past_key_value: Optional[Tuple[torch.Tensor, torch.Tensor]] = None, + use_cache: bool = False, + ) -> Tuple[torch.Tensor, Optional[Tuple[torch.Tensor, torch.Tensor]], torch.Tensor]: + # Gradient checkpointing: recompute forward in backward pass to save VRAM + if self.gradient_checkpointing and self.training: + return self._forward_checkpoint( + hidden_states, attention_mask, position_ids, past_key_value, use_cache, + ) + return self._forward( + hidden_states, attention_mask, position_ids, past_key_value, use_cache, + ) + + def _forward( + self, + hidden_states: torch.Tensor, + attention_mask: Optional[torch.Tensor] = None, + position_ids: Optional[torch.Tensor] = None, + past_key_value: Optional[Tuple[torch.Tensor, torch.Tensor]] = None, + use_cache: bool = False, + ) -> Tuple[torch.Tensor, Optional[Tuple[torch.Tensor, torch.Tensor]], torch.Tensor]: + residual = hidden_states + + # Pre-norm + Self-attention + hidden_states = self.input_norm(hidden_states) + attn_output, new_kv = self.self_attn( + hidden_states, + attention_mask=attention_mask, + position_ids=position_ids, + past_key_value=past_key_value, + use_cache=use_cache, + ) + hidden_states = residual + attn_output + + # Pre-norm + MoE FFN (v0.4 fix: forward attention_mask for proper aux loss) + residual = hidden_states + hidden_states = self.post_attention_norm(hidden_states) + moe_output, aux_loss = self.moe(hidden_states, attention_mask=attention_mask) + hidden_states = residual + moe_output + + return hidden_states, new_kv, aux_loss + + def _forward_checkpoint( + self, + hidden_states: torch.Tensor, + attention_mask: Optional[torch.Tensor] = None, + position_ids: Optional[torch.Tensor] = None, + past_key_value: Optional[Tuple[torch.Tensor, torch.Tensor]] = None, + use_cache: bool = False, + ) -> Tuple[torch.Tensor, Optional[Tuple[torch.Tensor, torch.Tensor]], torch.Tensor]: + """Gradient checkpointing wrapper — recompute forward in backward pass.""" + def custom_forward(*inputs): + return self._forward(*inputs) + + layers_outputs = torch.utils.checkpoint.checkpoint( + custom_forward, + hidden_states, + attention_mask, + position_ids, + past_key_value, + use_cache, + use_reentrant=False, + ) + return layers_outputs diff --git a/nexus/optim/__init__.py b/nexus/optim/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..b82a01a4a91ec1a3d34f1487e1e9d5bb8aabe86b --- /dev/null +++ b/nexus/optim/__init__.py @@ -0,0 +1,25 @@ +""" +Nexus Optim Module - v0.2 NEW +============================= +Tối ưu model cho inference và training. + +Modules: +- quantization: INT8/INT4/FP8 quantization +- lora: LoRA / QLoRA efficient fine-tuning +- distillation: Knowledge distillation +- pruning: Structured / unstructured pruning +""" + +from .quantization import Quantizer +from .lora import LoRAConfig, LoRALinear, apply_lora +from .distillation import Distiller +from .pruning import Pruner + +__all__ = [ + "Quantizer", + "LoRAConfig", + "LoRALinear", + "apply_lora", + "Distiller", + "Pruner", +] diff --git a/nexus/optim/distillation.py b/nexus/optim/distillation.py new file mode 100644 index 0000000000000000000000000000000000000000..5583f9fad058f96c5765e2dc2556f64ee59a86e3 --- /dev/null +++ b/nexus/optim/distillation.py @@ -0,0 +1,145 @@ +"""Knowledge Distillation - Train small model từ large teacher.""" +from __future__ import annotations + +import torch +import torch.nn as nn +import torch.nn.functional as F +from typing import Optional, Dict, Callable, List +from dataclasses import dataclass +import logging + +logger = logging.getLogger(__name__) + + +@dataclass +class DistillationConfig: + """Config cho knowledge distillation.""" + temperature: float = 2.0 # Softmax temperature + alpha: float = 0.5 # Weight for distillation loss (1-alpha for hard labels) + hard_label_loss: str = "ce" # "ce", "focal", "label_smoothing" + label_smoothing: float = 0.1 + teacher_temp: Optional[float] = None # Defaults to temperature + + +class Distiller: + """Knowledge distillation: train student model from teacher. + + Loss = α * KL(teacher_soft || student_soft) * T² + + (1-α) * CE(student_hard, labels) + + Usage: + distiller = Distiller(config=DistillationConfig(temperature=4.0)) + for batch in dataloader: + loss = distiller.compute_loss( + student_logits=student(batch), + teacher_logits=teacher(batch), # no_grad + labels=batch_labels, + ) + loss.backward() + """ + + def __init__(self, config: DistillationConfig = None): + self.config = config or DistillationConfig() + + def compute_loss( + self, + student_logits: torch.Tensor, + teacher_logits: torch.Tensor, + labels: Optional[torch.Tensor] = None, + ) -> Dict[str, torch.Tensor]: + """Compute distillation loss. + + Args: + student_logits: [B, V] logits from student model + teacher_logits: [B, V] logits from teacher model (should be no_grad) + labels: [B] ground truth labels (optional, for hard label loss) + + Returns: + Dict with 'loss', 'distill_loss', 'hard_loss' tensors + """ + cfg = self.config + T = cfg.temperature + teacher_T = cfg.teacher_temp or T + + # Distillation loss: KL divergence between soft predictions + student_log_probs = F.log_softmax(student_logits / T, dim=-1) + teacher_probs = F.softmax(teacher_logits / teacher_T, dim=-1) + + # KL(teacher || student) = sum(teacher * log(teacher/student)) + # = sum(teacher * log(teacher)) - sum(teacher * log(student)) + # We only need the second term (first is constant w.r.t. student) + kl_loss = -(teacher_probs * student_log_probs).sum(dim=-1).mean() + # Scale by T² (per Hinton et al.) + distill_loss = kl_loss * (T ** 2) + + # Hard label loss + hard_loss = torch.tensor(0.0, device=student_logits.device) + if labels is not None: + if cfg.hard_label_loss == "ce": + hard_loss = F.cross_entropy(student_logits, labels) + elif cfg.hard_label_loss == "focal": + # Focal loss + ce = F.cross_entropy(student_logits, labels, reduction="none") + pt = torch.exp(-ce) + hard_loss = ((1 - pt) ** 2 * ce).mean() + elif cfg.hard_label_loss == "label_smoothing": + hard_loss = F.cross_entropy( + student_logits, labels, + label_smoothing=cfg.label_smoothing, + ) + + # Total loss + total_loss = cfg.alpha * distill_loss + (1 - cfg.alpha) * hard_loss + + return { + "loss": total_loss, + "distill_loss": distill_loss, + "hard_loss": hard_loss, + } + + def train_step( + self, + student: nn.Module, + teacher: nn.Module, + batch: Dict[str, torch.Tensor], + optimizer: torch.optim.Optimizer, + ) -> Dict[str, float]: + """One distillation training step. + + Args: + student: Student model (trainable) + teacher: Teacher model (will be set to eval, no_grad) + batch: Dict with 'input_ids', 'attention_mask', 'labels' + optimizer: Optimizer for student + + Returns: + Dict of loss values + """ + teacher.eval() + + with torch.no_grad(): + teacher_outputs = teacher( + input_ids=batch["input_ids"], + attention_mask=batch.get("attention_mask"), + ) + teacher_logits = teacher_outputs["logits"] if isinstance(teacher_outputs, dict) else teacher_outputs + + student.train() + student_outputs = student( + input_ids=batch["input_ids"], + attention_mask=batch.get("attention_mask"), + ) + student_logits = student_outputs["logits"] if isinstance(student_outputs, dict) else student_outputs + + losses = self.compute_loss( + student_logits=student_logits, + teacher_logits=teacher_logits, + labels=batch.get("labels"), + ) + + optimizer.zero_grad() + losses["loss"].backward() + torch.nn.utils.clip_grad_norm_(student.parameters(), 1.0) + optimizer.step() + + return {k: v.item() for k, v in losses.items()} diff --git a/nexus/optim/lora.py b/nexus/optim/lora.py new file mode 100644 index 0000000000000000000000000000000000000000..9995b8c5a5bfab09a220cd931d85db374c004b79 --- /dev/null +++ b/nexus/optim/lora.py @@ -0,0 +1,151 @@ +"""LoRA - Low-Rank Adaptation cho efficient fine-tuning.""" +from __future__ import annotations + +import math +import torch +import torch.nn as nn +from typing import Dict, List, Optional, Set +from dataclasses import dataclass, field + + +@dataclass +class LoRAConfig: + """Config cho LoRA.""" + rank: int = 8 # LoRA rank (r) + alpha: int = 16 # LoRA scaling factor (α) + dropout: float = 0.0 # LoRA dropout + # v0.4 fix: thay "gate_proj"+"up_proj" → "gate_up_proj" vì v0.3 SwiGLU(parallel=True) + # fuses gate+up thành 1 matmul. Nếu không có gate_up_proj, có thể truyền cả 3. + target_modules: List[str] = field(default_factory=lambda: [ + "q_proj", "k_proj", "v_proj", "o_proj", # attention + "gate_up_proj", "down_proj", # FFN (MLP-parallel) + ]) + bias: str = "none" # "none", "all", "lora_only" + modules_to_save: List[str] = field(default_factory=list) # Full-finetune these + fan_in_fan_out: bool = False + + @property + def scaling(self) -> float: + if self.rank <= 0: + return 0.0 + return self.alpha / self.rank + + +class LoRALinear(nn.Module): + """Linear layer với LoRA adaptation. + + Adds low-rank matrices A and B such that: + output = original(x) + scaling * B(A(x)) + + Only A and B are trainable; original weights are frozen. + """ + + def __init__( + self, + original: nn.Linear, + rank: int = 8, + alpha: int = 16, + dropout: float = 0.0, + ): + super().__init__() + self.original = original + self.rank = rank + self.alpha = alpha + self.scaling = alpha / rank + + # Freeze original + for param in self.original.parameters(): + param.requires_grad = False + + # LoRA matrices + in_features = original.in_features + out_features = original.out_features + + # A: in_features × rank (init with kaiming) + self.lora_A = nn.Parameter(torch.zeros(rank, in_features)) + nn.init.kaiming_uniform_(self.lora_A, a=math.sqrt(5)) + + # B: rank × out_features (init with zeros) + self.lora_B = nn.Parameter(torch.zeros(out_features, rank)) + + self.dropout = nn.Dropout(dropout) if dropout > 0 else nn.Identity() + + def forward(self, x: torch.Tensor) -> torch.Tensor: + # Original output + original_out = self.original(x) + # LoRA delta: x @ A^T @ B^T * scaling + lora_out = self.dropout(x) @ self.lora_A.T @ self.lora_B.T * self.scaling + return original_out + lora_out + + def merge(self) -> nn.Linear: + """Merge LoRA weights into original (for inference).""" + with torch.no_grad(): + delta = (self.lora_B @ self.lora_A) * self.scaling + self.original.weight.data += delta + return self.original + + def extra_repr(self) -> str: + return f"rank={self.rank}, alpha={self.alpha}, scaling={self.scaling:.3f}" + + +def apply_lora( + model: nn.Module, + config: LoRAConfig, +) -> nn.Module: + """Apply LoRA to a model. + + Replaces target Linear modules with LoRALinear. + Returns the modified model. + + Usage: + config = LoRAConfig(rank=8, target_modules=["q_proj", "v_proj"]) + model = apply_lora(model, config) + # Now only LoRA params are trainable + """ + target_modules = set(config.target_modules) + + def _replace_recursive(module: nn.Module, prefix: str = ""): + for name, child in list(module.named_children()): + full_name = f"{prefix}.{name}" if prefix else name + # Check if this module should be LoRA-adapted + short_name = name + if short_name in target_modules and isinstance(child, nn.Linear): + lora_layer = LoRALinear( + original=child, + rank=config.rank, + alpha=config.alpha, + dropout=config.dropout, + ) + setattr(module, name, lora_layer) + else: + _replace_recursive(child, full_name) + + _replace_recursive(model) + + # Make sure non-LoRA params are frozen + for name, param in model.named_parameters(): + if "lora_" not in name and name not in config.modules_to_save: + param.requires_grad = False + + return model + + +def get_lora_state_dict(model: nn.Module) -> Dict[str, torch.Tensor]: + """Get only LoRA params (for saving).""" + return { + name: param + for name, param in model.named_parameters() + if "lora_" in name and param.requires_grad + } + + +def count_lora_params(model: nn.Module) -> Dict[str, int]: + """Count trainable vs total params.""" + total = sum(p.numel() for p in model.parameters()) + trainable = sum(p.numel() for p in model.parameters() if p.requires_grad) + return { + "total": total, + "trainable": trainable, + "frozen": total - trainable, + "trainable_pct": trainable / total * 100 if total > 0 else 0, + } diff --git a/nexus/optim/pruning.py b/nexus/optim/pruning.py new file mode 100644 index 0000000000000000000000000000000000000000..f036e66082038a3a8fd1448b7bacd4525700b021 --- /dev/null +++ b/nexus/optim/pruning.py @@ -0,0 +1,160 @@ +"""Pruning - Structured/unstructured pruning.""" +from __future__ import annotations + +import torch +import torch.nn as nn +from typing import Dict, List, Optional, Tuple +from dataclasses import dataclass +import logging + +logger = logging.getLogger(__name__) + + +@dataclass +class PruningConfig: + """Config cho pruning.""" + method: str = "magnitude_unstructured" # "magnitude_unstructured", "magnitude_structured", "random" + amount: float = 0.2 # Fraction of weights to prune (0.0-1.0) + target_modules: List[str] = None # Default: all Linear + dim: int = 0 # For structured: which dim to prune + n_prune_steps: int = 1 # Iterative pruning steps + + +class Pruner: + """Prune model weights để giảm params và inference cost. + + Methods: + - magnitude_unstructured: Prune smallest-magnitude weights (set to 0) + - magnitude_structured: Remove entire neurons/channels + - random: Random pruning (baseline) + + Usage: + pruner = Pruner(config=PruningConfig(amount=0.3)) + pruned_model = pruner.prune(model) + """ + + def __init__(self, config: PruningConfig = None): + self.config = config or PruningConfig() + if self.config.target_modules is None: + self.config.target_modules = [nn.Linear] + + def prune(self, model: nn.Module) -> nn.Module: + """Prune model in-place.""" + method = self.config.method + + if method == "magnitude_unstructured": + return self._prune_magnitude_unstructured(model) + elif method == "magnitude_structured": + return self._prune_magnitude_structured(model) + elif method == "random": + return self._prune_random(model) + else: + raise ValueError(f"Unknown pruning method: {method}") + + def _prune_magnitude_unstructured(self, model: nn.Module) -> nn.Module: + """Prune smallest-magnitude weights (set to 0).""" + try: + from torch.nn.utils import prune + except ImportError: + logger.error("torch.nn.utils.prune not available") + return model + + amount = self.config.amount + + for name, module in model.named_modules(): + if isinstance(module, tuple(self.config.target_modules)): + prune.l1_unstructured(module, name="weight", amount=amount) + # Make pruning permanent + prune.remove(module, "weight") + + # Count sparsity + sparsity = self._compute_sparsity(model) + logger.info(f"Magnitude unstructured pruning: {sparsity*100:.1f}% weights pruned") + return model + + def _prune_magnitude_structured(self, model: nn.Module) -> nn.Module: + """Remove entire neurons/channels based on L2 norm.""" + try: + from torch.nn.utils import prune + except ImportError: + logger.error("torch.nn.utils.prune not available") + return model + + amount = self.config.amount + dim = self.config.dim + + for name, module in model.named_modules(): + if isinstance(module, tuple(self.config.target_modules)): + prune.ln_structured(module, name="weight", amount=amount, n=2, dim=dim) + prune.remove(module, "weight") + + sparsity = self._compute_sparsity(model) + logger.info(f"Magnitude structured pruning (dim={dim}): {sparsity*100:.1f}% pruned") + return model + + def _prune_random(self, model: nn.Module) -> nn.Module: + """Random pruning (baseline).""" + try: + from torch.nn.utils import prune + except ImportError: + return model + + amount = self.config.amount + + for name, module in model.named_modules(): + if isinstance(module, tuple(self.config.target_modules)): + prune.random_unstructured(module, name="weight", amount=amount) + prune.remove(module, "weight") + + return model + + def _compute_sparsity(self, model: nn.Module) -> float: + """Compute fraction of zero weights.""" + total = 0 + zeros = 0 + for param in model.parameters(): + total += param.numel() + zeros += (param == 0).sum().item() + return zeros / total if total > 0 else 0 + + def iterative_prune( + self, + model: nn.Module, + train_fn=None, + steps: int = None, + ) -> nn.Module: + """Iterative pruning: prune, retrain, prune, retrain, ... + + Args: + model: Model to prune + train_fn: Function(model) to retrain after each prune step + steps: Number of prune-retrain cycles (default: config.n_prune_steps) + """ + steps = steps or self.config.n_prune_steps + amount_per_step = self.config.amount / steps + + original_config = self.config.amount + self.config.amount = amount_per_step + + for step in range(steps): + logger.info(f"Iterative pruning step {step+1}/{steps}") + self.prune(model) + if train_fn: + logger.info("Retraining after pruning...") + train_fn(model) + + self.config.amount = original_config + return model + + def stats(self, model: nn.Module) -> Dict[str, float]: + """Get pruning stats.""" + sparsity = self._compute_sparsity(model) + total_params = sum(p.numel() for p in model.parameters()) + nonzero_params = sum((p != 0).sum().item() for p in model.parameters()) + return { + "total_params": total_params, + "nonzero_params": nonzero_params, + "zero_params": total_params - nonzero_params, + "sparsity": sparsity, + "compression_ratio": 1 / (1 - sparsity) if sparsity < 1 else float("inf"), + } diff --git a/nexus/optim/quantization.py b/nexus/optim/quantization.py new file mode 100644 index 0000000000000000000000000000000000000000..17513d5cbd898c74fa723dcaeed25d3c93509981 --- /dev/null +++ b/nexus/optim/quantization.py @@ -0,0 +1,168 @@ +"""Quantization - INT8/INT4/FP8 quantization cho model.""" +from __future__ import annotations + +import torch +import torch.nn as nn +from typing import Dict, Any, Optional, Tuple +from dataclasses import dataclass +import logging + +logger = logging.getLogger(__name__) + + +@dataclass +class QuantizationConfig: + """Config cho quantization.""" + method: str = "int8" # "int8", "int4", "fp8" + granularity: str = "per_channel" # "per_tensor", "per_channel" + calibration_samples: int = 128 + calibration_batches: int = 4 + skip_layers: list = None # Layers to skip quantization + + def __post_init__(self): + if self.skip_layers is None: + self.skip_layers = ["lm_head", "embed_tokens"] + + +class Quantizer: + """Quantize model weights để giảm memory footprint. + + Supported methods: + - INT8: 4x memory reduction, minimal quality loss + - INT4: 8x memory reduction, slight quality loss + - FP8: 2x memory reduction, almost no quality loss (H100 only) + + Usage: + quantizer = Quantizer(config=QuantizationConfig(method="int8")) + quantized_model = quantizer.quantize(model, calibration_data) + """ + + def __init__(self, config: QuantizationConfig = None): + self.config = config or QuantizationConfig() + + def quantize( + self, + model: nn.Module, + calibration_data: Optional[torch.Tensor] = None, + ) -> nn.Module: + """Quantize model in-place. + + Args: + model: Model to quantize + calibration_data: Sample inputs for activation calibration + + Returns: + Quantized model (same object, modified in-place) + """ + method = self.config.method + + if method == "int8": + return self._quantize_int8(model, calibration_data) + elif method == "int4": + return self._quantize_int4(model, calibration_data) + elif method == "fp8": + return self._quantize_fp8(model, calibration_data) + else: + raise ValueError(f"Unknown quantization method: {method}") + + def _quantize_int8( + self, + model: nn.Module, + calibration_data: Optional[torch.Tensor], + ) -> nn.Module: + """Quantize to INT8 using PyTorch dynamic quantization.""" + # Use PyTorch built-in dynamic quantization + # Works on Linear layers + quantized = torch.quantization.quantize_dynamic( + model, + {nn.Linear}, + dtype=torch.qint8, + ) + logger.info(f"INT8 quantization done. Memory reduced ~2x.") + return quantized + + def _quantize_int4( + self, + model: nn.Module, + calibration_data: Optional[torch.Tensor], + ) -> nn.Module: + """Quantize to INT4 (requires bitsandbytes library).""" + try: + import bitsandbytes as bnb + except ImportError: + logger.warning( + "bitsandbytes not installed. Install with: pip install bitsandbytes. " + "Falling back to INT8." + ) + return self._quantize_int8(model, calibration_data) + + # Replace Linear layers with INT4 versions + for name, module in model.named_children(): + if isinstance(module, nn.Linear) and name not in self.config.skip_layers: + new_module = bnb.nn.Linear4bit( + module.in_features, + module.out_features, + bias=module.bias is not None, + compute_dtype=torch.float16, + ) + setattr(model, name, new_module) + elif hasattr(module, "children"): + self._quantize_int4(module, calibration_data) + + logger.info("INT4 quantization done. Memory reduced ~4x.") + return model + + def _quantize_fp8( + self, + model: nn.Module, + calibration_data: Optional[torch.Tensor], + ) -> nn.Module: + """Quantize to FP8 (requires H100 GPU or newer).""" + if not torch.cuda.is_available(): + logger.warning("FP8 requires CUDA. Falling back to INT8.") + return self._quantize_int8(model, calibration_data) + + capability = torch.cuda.get_device_capability() + if capability[0] < 9: + logger.warning(f"FP8 requires H100 (compute capability 9.0+). Got {capability}. Falling back to INT8.") + return self._quantize_int8(model, calibration_data) + + # FP8 conversion (when torch supports it natively) + try: + # v0.4 fix: skip_layers should match either "name." OR "name" prefix. + skip_set = set(self.config.skip_layers) + # Convert model to float8_e4m3fn + for name, param in model.named_parameters(): + # Skip if name starts with any skip layer prefix + if any( + name == s or name.startswith(s + ".") or name.startswith(s) + for s in skip_set + ): + continue + # Also skip embeddings/lm_head typically + if "embed_tokens" in name or "lm_head" in name: + continue + param.data = param.data.to(torch.float8_e4m3fn) + logger.info("FP8 quantization done. Memory reduced ~2x.") + except Exception as e: + logger.warning(f"FP8 conversion failed: {e}. Falling back to INT8.") + return self._quantize_int8(model, calibration_data) + + return model + + def estimate_memory_savings(self, model: nn.Module) -> Dict[str, float]: + """Estimate memory savings.""" + total_params = sum(p.numel() for p in model.parameters()) + fp16_mb = (total_params * 2) / (1024 * 1024) + int8_mb = (total_params * 1) / (1024 * 1024) + int4_mb = (total_params * 0.5) / (1024 * 1024) + fp8_mb = (total_params * 1) / (1024 * 1024) + + return { + "fp16_mb": fp16_mb, + "int8_mb": int8_mb, + "int4_mb": int4_mb, + "fp8_mb": fp8_mb, + "int8_savings_pct": (1 - int8_mb / fp16_mb) * 100, + "int4_savings_pct": (1 - int4_mb / fp16_mb) * 100, + } diff --git a/nexus/safety/__init__.py b/nexus/safety/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..2226b4a7785577c8c82416b7c3043db84df74517 --- /dev/null +++ b/nexus/safety/__init__.py @@ -0,0 +1,18 @@ +"""Nexus Safety Module - v0.2 NEW; v0.4 fix: expose get_default_guardrails.""" +from .filters import SafetyFilter, ContentFilter, PIIFilter +from .guardrails import ( + Guardrail, + GuardrailAction, + GuardrailManager, + get_default_guardrails, +) + +__all__ = [ + "SafetyFilter", + "ContentFilter", + "PIIFilter", + "Guardrail", + "GuardrailAction", + "GuardrailManager", + "get_default_guardrails", +] diff --git a/nexus/safety/filters.py b/nexus/safety/filters.py new file mode 100644 index 0000000000000000000000000000000000000000..71c5c5a20ea7cfea87ad6e1e0bd43e6b80e0f4c0 --- /dev/null +++ b/nexus/safety/filters.py @@ -0,0 +1,144 @@ +"""Safety Filters - Lọc nội dung không an toàn.""" +from __future__ import annotations + +import re +from typing import Dict, Any, List, Optional, Tuple +from dataclasses import dataclass + + +@dataclass +class FilterResult: + """Kết quả filter.""" + passed: bool + score: float # 0.0 = unsafe, 1.0 = safe + reason: Optional[str] = None + categories: List[str] = None + + +class ContentFilter: + """Filter nội dung toxic / harmful.""" + + HARMFUL_PATTERNS = [ + # Violence + (r"\b(kill|murder|assassinate|execute)\s+(someone|him|her|them|people)\b", "violence"), + (r"\bbomb\s+(recipe|how\s+to\s+make)\b", "violence"), + # Hate speech patterns + (r"\b(racial|ethnic)\s+slur\b", "hate_speech"), + # Self-harm + (r"\b(suicide|self-harm)\s+(method|how\s+to)\b", "self_harm"), + # Illegal + (r"\b(drug|cocaine|heroin)\s+(recipe|manufacture|synthesize)\b", "illegal"), + (r"\bchild\s+exploitation\b", "illegal"), + ] + + def __init__(self): + self._compiled = [(re.compile(p, re.IGNORECASE), cat) for p, cat in self.HARMFUL_PATTERNS] + + def check(self, text: str) -> FilterResult: + """Check text for harmful content.""" + if not text: + return FilterResult(passed=True, score=1.0) + + matched_categories = [] + for pattern, category in self._compiled: + if pattern.search(text): + matched_categories.append(category) + + if matched_categories: + return FilterResult( + passed=False, + score=0.0, + reason=f"Harmful content detected: {', '.join(matched_categories)}", + categories=matched_categories, + ) + + return FilterResult(passed=True, score=1.0) + + +class PIIFilter: + """Detect and mask PII (Personally Identifiable Information).""" + + PII_PATTERNS = { + "email": re.compile(r"\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Z|a-z]{2,}\b"), + "phone": re.compile(r"\b\+?[\d\s\-\(\)]{10,15}\b"), + "ssn": re.compile(r"\b\d{3}-\d{2}-\d{4}\b"), + "credit_card": re.compile(r"\b(?:\d{4}[\s\-]?){3}\d{4}\b"), + "ip": re.compile(r"\b\d{1,3}\.\d{1,3}\.\d{1,3}\.\d{1,3}\b"), + "api_key": re.compile(r"\b(?:sk-|pk-|ghp_|gho_|github_pat_)[A-Za-z0-9]{20,}\b"), + } + + def check(self, text: str) -> FilterResult: + """Check for PII.""" + if not text: + return FilterResult(passed=True, score=1.0) + + found = [] + for pii_type, pattern in self.PII_PATTERNS.items(): + if pattern.search(text): + found.append(pii_type) + + if found: + return FilterResult( + passed=False, + score=0.3, + reason=f"PII detected: {', '.join(found)}", + categories=found, + ) + + return FilterResult(passed=True, score=1.0) + + def mask(self, text: str) -> str: + """Mask PII in text.""" + for pii_type, pattern in self.PII_PATTERNS.items(): + text = pattern.sub(f"[{pii_type.upper()}_REDACTED]", text) + return text + + +class SafetyFilter: + """Composite safety filter.""" + + def __init__(self, enable_content: bool = True, enable_pii: bool = True): + self.content_filter = ContentFilter() if enable_content else None + self.pii_filter = PIIFilter() if enable_pii else None + + def check(self, text: str) -> FilterResult: + """Run all filters.""" + results = [] + if self.content_filter: + results.append(("content", self.content_filter.check(text))) + if self.pii_filter: + results.append(("pii", self.pii_filter.check(text))) + + if not results: + return FilterResult(passed=True, score=1.0) + + # Aggregate: passed only if ALL pass + all_passed = all(r.passed for _, r in results) + min_score = min(r.score for _, r in results) + + if all_passed: + return FilterResult(passed=True, score=min_score) + + reasons = [f"{name}: {r.reason}" for name, r in results if not r.passed] + categories = [] + for _, r in results: + if r.categories: + categories.extend(r.categories) + + return FilterResult( + passed=False, + score=min_score, + reason="; ".join(reasons), + categories=categories, + ) + + def sanitize(self, text: str) -> Tuple[str, FilterResult]: + """Sanitize text: check + mask PII.""" + result = self.check(text) + if not result.passed and self.pii_filter: + # Try masking PII + masked = self.pii_filter.mask(text) + recheck = self.check(masked) + if recheck.passed: + return masked, recheck + return text, result diff --git a/nexus/safety/guardrails.py b/nexus/safety/guardrails.py new file mode 100644 index 0000000000000000000000000000000000000000..2cdb5c4bc7c43d794284ab195a279ffe0413387e --- /dev/null +++ b/nexus/safety/guardrails.py @@ -0,0 +1,161 @@ +"""Guardrails - Bảo vệ model khỏi misuse.""" +from __future__ import annotations + +from typing import Dict, Any, List, Optional, Callable +from dataclasses import dataclass, field +from enum import Enum + + +class GuardrailAction(str, Enum): + """Action khi guardrail trigger.""" + ALLOW = "allow" + WARN = "warn" + BLOCK = "block" + REDACT = "redact" + + +@dataclass +class Guardrail: + """Một guardrail rule.""" + name: str + description: str + check_fn: Callable[[str], bool] + action: GuardrailAction = GuardrailAction.BLOCK + message: str = "" + + def evaluate(self, text: str) -> Dict[str, Any]: + triggered = self.check_fn(text) + return { + "name": self.name, + "triggered": triggered, + "action": self.action.value if triggered else GuardrailAction.ALLOW.value, + "message": self.message if triggered else "", + } + + +class GuardrailManager: + """Quản lý nhiều guardrails. + + Usage: + mgr = GuardrailManager() + mgr.add(Guardrail( + name="no_secrets", + description="Block API keys", + check_fn=lambda t: "sk-" in t or "ghp_" in t, + action=GuardrailAction.BLOCK, + message="API keys are not allowed", + )) + result = mgr.check(user_input) + if not result["allowed"]: + print(result["message"]) + """ + + def __init__(self): + self._guardrails: List[Guardrail] = [] + + def add(self, guardrail: Guardrail) -> None: + """Add a guardrail.""" + self._guardrails.append(guardrail) + + def remove(self, name: str) -> Optional[Guardrail]: + """Remove a guardrail by name.""" + for i, g in enumerate(self._guardrails): + if g.name == name: + return self._guardrails.pop(i) + return None + + def check(self, text: str) -> Dict[str, Any]: + """Check text against all guardrails. + + Returns: + Dict with: + - allowed: bool + - triggered: List of triggered guardrail names + - action: overall action (most restrictive) + - message: combined messages + """ + triggered = [] + actions = [] + messages = [] + + for g in self._guardrails: + result = g.evaluate(text) + if result["triggered"]: + triggered.append(g.name) + actions.append(g.action) + if g.message: + messages.append(g.message) + + # Most restrictive action + action_priority = { + GuardrailAction.BLOCK: 4, + GuardrailAction.REDACT: 3, + GuardrailAction.WARN: 2, + GuardrailAction.ALLOW: 1, + } + + if not actions: + overall_action = GuardrailAction.ALLOW + else: + overall_action = max(actions, key=lambda a: action_priority.get(a, 0)) + + return { + "allowed": overall_action in (GuardrailAction.ALLOW, GuardrailAction.WARN), + "triggered": triggered, + "action": overall_action.value, + "message": "; ".join(messages) if messages else "", + } + + def list_guardrails(self) -> List[Dict[str, str]]: + """List all registered guardrails.""" + return [ + { + "name": g.name, + "description": g.description, + "action": g.action.value, + } + for g in self._guardrails + ] + + +def get_default_guardrails() -> GuardrailManager: + """Get default guardrail configuration.""" + mgr = GuardrailManager() + + # No secrets + mgr.add(Guardrail( + name="no_api_keys", + description="Block obvious API keys and tokens", + check_fn=lambda t: any(s in t for s in ["sk-", "ghp_", "gho_", "github_pat_", "AKIA"]), + action=GuardrailAction.BLOCK, + message="API keys/tokens are not allowed in input", + )) + + # No PII + mgr.add(Guardrail( + name="no_pii", + description="Warn on PII (email, phone, SSN)", + check_fn=lambda t: any(c in t for c in ["@", "ssn", "social security"]), + action=GuardrailAction.WARN, + message="PII detected - please remove personal information", + )) + + # No harmful content + mgr.add(Guardrail( + name="no_harmful", + description="Block harmful content (violence, illegal)", + check_fn=lambda t: any(w in t.lower() for w in ["bomb recipe", "kill tutorial", "drug manufacture"]), + action=GuardrailAction.BLOCK, + message="Harmful content is not allowed", + )) + + # Length limit + mgr.add(Guardrail( + name="length_limit", + description="Warn on very long inputs", + check_fn=lambda t: len(t) > 50000, + action=GuardrailAction.WARN, + message="Input is very long, may be truncated", + )) + + return mgr diff --git a/nexus/skills/__init__.py b/nexus/skills/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..2230ac6961241ef4e6b767321749dfb2e7adb343 --- /dev/null +++ b/nexus/skills/__init__.py @@ -0,0 +1,39 @@ +""" +Nexus Skills Module - v0.2 NEW +============================== +Hệ thống Skills cho Nexus Coder Agent. + +Skills là các năng lực chuyên môn được tổ chức theo domain: +- code_generation: Sinh code từ spec +- code_review: Review code, tìm bugs +- code_refactor: Tái cấu trúc code +- debugging: Debug và fix lỗi +- documentation: Sinh docs +- testing: Sinh unit tests +- algorithm_design: Thiết kế thuật toán +- data_analysis: Phân tích dữ liệu +- translation: Dịch Việt-Anh +- summarization: Tóm tắt văn bản +- reasoning: Suy luận logic +- math_skill: Giải toán +- sql_generation: Sinh SQL +- security_audit: Audit bảo mật +- performance_opt: Tối ưu hiệu năng + +Usage: + from nexus.skills import SkillRegistry + registry = SkillRegistry() + skill = registry.get("code_generation") + result = skill.execute(prompt="viết hàm sort", context={}) +""" + +from .base import Skill, SkillResult, SkillContext +from .registry import SkillRegistry, get_global_registry + +__all__ = [ + "Skill", + "SkillResult", + "SkillContext", + "SkillRegistry", + "get_global_registry", +] diff --git a/nexus/skills/algorithm_design.py b/nexus/skills/algorithm_design.py new file mode 100644 index 0000000000000000000000000000000000000000..8e14bb235d76a341916e8b2e0f006fa909ab262d --- /dev/null +++ b/nexus/skills/algorithm_design.py @@ -0,0 +1,61 @@ +"""Algorithm Design Skill - Thiết kế thuật toán.""" +from __future__ import annotations +from typing import List +from .base import Skill, SkillResult, SkillContext, SkillCategory, SkillPriority + + +class AlgorithmDesignSkill(Skill): + """Thiết kế thuật toán: complexity analysis, optimization, data structure selection.""" + + category = SkillCategory.REASONING + priority = SkillPriority.MEDIUM + keywords: List[str] = [ + "algorithm", "thuật toán", "complexity", "độ phức tạp", + "big o", "big-o", "optimize", "tối ưu", + "data structure", "cấu trúc dữ liệu", "sort", "sắp xếp", + "search", "tìm kiếm", "graph", "đồ thị", "tree", "cây", + "dynamic programming", "quy hoạch động", + ] + + @property + def name(self) -> str: + return "algorithm_design" + + @property + def description(self) -> str: + return ( + "Thiết kế thuật toán: chọn data structure, phân tích complexity, " + "optimize time/space, so sánh approaches, implement clean." + ) + + def execute(self, context: SkillContext) -> SkillResult: + approaches = [ + "Brute force (baseline)", + "Greedy algorithm", + "Divide and conquer", + "Dynamic programming", + "Backtracking", + "Branch and bound", + "Graph algorithms (BFS, DFS, Dijkstra, A*)", + "Two pointers / Sliding window", + "Binary search", + "Monotonic stack / queue", + "Topological sort", + "Union-Find (Disjoint Set)", + "Segment tree / Fenwick tree", + "Sparse table", + ] + return SkillResult( + success=True, + output=f"[AlgorithmDesign] Considering {len(approaches)} approaches.", + metadata={ + "skill": self.name, + "approaches": approaches, + "complexity_targets": ["O(1)", "O(log n)", "O(n)", "O(n log n)", "O(n²)"], + }, + suggestions=[ + "Start with brute force, then optimize", + "Analyze time AND space complexity", + "Consider edge cases (empty, single, large inputs)", + ], + ) diff --git a/nexus/skills/anomaly_detection.py b/nexus/skills/anomaly_detection.py new file mode 100644 index 0000000000000000000000000000000000000000..69a03a75872ff7bf4e465fc0c62a0e22969d1cf1 --- /dev/null +++ b/nexus/skills/anomaly_detection.py @@ -0,0 +1,205 @@ +"""Anomaly Detection Skill - IsolationForest / LOF / DBSCAN / statistical. + +Sinh code phát hiện bất thường trong dữ liệu: Isolation Forest, Local Outlier +Factor (LOF), One-Class SVM, DBSCAN, Z-score & IQR rule, với visualisation +(PCA scatter + outlier highlight) và đánh giá (precision/recall nếu có ground truth). + +Author: Hieu Louis (2026) +""" +from __future__ import annotations + +from typing import Dict, List + +from .base import Skill, SkillCategory, SkillContext, SkillPriority, SkillResult + + +ANOMALY_CODE = '''"""Anomaly detection toolkit / Bộ phát hiện bất thường.""" +from __future__ import annotations +from typing import Dict +import numpy as np +import pandas as pd +from sklearn.ensemble import IsolationForest +from sklearn.neighbors import LocalOutlierFactor +from sklearn.svm import OneClassSVM +from sklearn.cluster import DBSCAN +from sklearn.preprocessing import StandardScaler +from sklearn.decomposition import PCA +from sklearn.metrics import precision_recall_fscore_support + + +def iqr_outliers(x: np.ndarray, k: float = 1.5) -> np.ndarray: + """IQR rule / Quy tắc IQR.""" + q1, q3 = np.percentile(x, [25, 75]) + iqr = q3 - q1 + mask = (x < q1 - k * iqr) | (x > q3 + k * iqr) + return mask + + +def zscore_outliers(x: np.ndarray, threshold: float = 3.0) -> np.ndarray: + """Z-score rule / Quy tắc Z-score.""" + return np.abs((x - x.mean()) / (x.std(ddof=0) + 1e-9)) > threshold + + +def isolation_forest( + X: np.ndarray, contamination: float = 0.05, random_state: int = 42, +) -> Dict[str, object]: + """Isolation Forest — mạnh với high-dimensional, không cần assumption.""" + model = IsolationForest( + n_estimators=300, max_samples="auto", + contamination=contamination, random_state=random_state, n_jobs=-1, + ) + labels = model.fit_predict(X) # 1=inlier, -1=outlier + scores = -model.decision_function(X) # higher = more anomalous + return {"model": model, "labels": labels, "scores": scores} + + +def local_outlier_factor(X: np.ndarray, n_neighbors: int = 20) -> Dict[str, object]: + """LOF — phát hiện anomaly dựa trên mật độ cục bộ.""" + lof = LocalOutlierFactor(n_neighbors=n_neighbors, contamination="auto", n_jobs=-1) + labels = lof.fit_predict(X) + scores = -lof.negative_outlier_factor_ + return {"labels": labels, "scores": scores} + + +def one_class_svm(X: np.ndarray, nu: float = 0.05) -> Dict[str, object]: + """One-Class SVM — tốt cho novelty detection khi train chỉ có normal.""" + scaler = StandardScaler().fit(X) + Xs = scaler.transform(X) + model = OneClassSVM(kernel="rbf", gamma="scale", nu=nu) + labels = model.fit_predict(Xs) + scores = -model.decision_function(Xs) + return {"model": model, "scaler": scaler, "labels": labels, "scores": scores} + + +def dbscan_outliers(X: np.ndarray, eps: float = 0.5, min_samples: int = 5) -> np.ndarray: + """DBSCAN — điểm không thuộc cụm nào (-1) là anomaly.""" + db = DBSCAN(eps=eps, min_samples=min_samples, n_jobs=-1).fit(StandardScaler().fit_transform(X)) + return db.labels_ == -1 + + +def evaluate(y_true: np.ndarray, y_pred: np.ndarray) -> Dict[str, float]: + """Đánh giá khi có ground-truth (-1 = outlier, 1 = inlier).""" + p, r, f, _ = precision_recall_fscore_support( + y_true, y_pred, average="binary", pos_label=-1, zero_division=0, + ) + return {"precision": float(p), "recall": float(r), "f1": float(f)} + + +def visualize(X: np.ndarray, labels: np.ndarray, title: str = "Anomalies"): + """PCA scatter 2D với outliers highlight / Vẽ PCA 2D.""" + import matplotlib.pyplot as plt + X2 = PCA(n_components=2).fit_transform(StandardScaler().fit_transform(X)) + plt.figure(figsize=(8, 5)) + plt.scatter(X2[labels == 1, 0], X2[labels == 1, 1], s=8, c="steelblue", label="inlier") + plt.scatter(X2[labels == -1, 0], X2[labels == -1, 1], s=18, c="crimson", label="anomaly") + plt.title(title) + plt.legend() + plt.tight_layout() + return plt.gcf() + + +if __name__ == "__main__": + rng = np.random.default_rng(7) + X = rng.normal(size=(1000, 4)) + X[:20] += rng.normal(5, 1, size=(20, 4)) # inject 20 anomalies + res = isolation_forest(X, contamination=0.05) + print("outliers detected:", (res["labels"] == -1).sum()) +''' + +MODEL_SELECTION = """ +Anomaly Detection — Model Selection Guide / Hướng dẫn chọn mô hình +================================================================== +| Method | Best For | Notes | +|-------------------|-----------------------------------|------------------------------------| +| IsolationForest | High-D, mixed-type, scalable | Default first choice | +| LOF (kNN-density) | Local anomalies, low-D | Slow for large N (O(n²)) | +| One-Class SVM | Novelty detection (train=normal) | Sensitive to scaling & gamma | +| DBSCAN | Cluster-based anomalies | Needs eps tuning (k-distance plot)| +| Z-score / IQR | Univariate, explainable baseline | Fails on multi-modal distributions| +| Autoencoder | Non-linear high-D, large data | Requires deep-learning infra | + +Contamination: prior estimate of anomaly ratio. Tune via: + - Domain knowledge (e.g. fraud rate = 0.1%) + - Top-K approach: take top-k highest scores as anomalies + - Score histogram: visual knee / elbow in distribution +""" + + +class AnomalyDetectionSkill(Skill): + """Sinh toolkit phát hiện anomaly cho tabular data.""" + + category = SkillCategory.ML + priority = SkillPriority.MEDIUM + keywords: List[str] = [ + "anomaly", "anomalies", "outlier", "outliers", "isolation forest", + "isolationforest", "lof", "local outlier", "one-class svm", + "dbscan", "z-score", "iqr", "novelty detection", "fraud", + ] + examples = [ + "Detect anomalies in sensor data với IsolationForest", + "Phát hiện outlier dùng LOF", + "Setup fraud detection pipeline", + ] + + @property + def name(self) -> str: + return "anomaly_detection" + + @property + def description(self) -> str: + return ( + "Sinh code phát hiện bất thường: IsolationForest, LOF, One-Class SVM, " + "DBSCAN, Z-score/IQR baseline + PCA visualization + evaluation." + ) + + def can_handle(self, prompt: str, context: SkillContext = None) -> float: + prompt_lower = prompt.lower() + score = 0.0 + for kw in self.keywords: + if kw in prompt_lower: + score += 0.14 + return min(1.0, score) + + def execute(self, context: SkillContext) -> SkillResult: + prompt_lower = (context.prompt or "").lower() + if "lof" in prompt_lower: + recommended = "lof" + elif "svm" in prompt_lower: + recommended = "one_class_svm" + elif "dbscan" in prompt_lower: + recommended = "dbscan" + elif "iqr" in prompt_lower or "z-score" in prompt_lower or "zscore" in prompt_lower: + recommended = "statistical" + else: + recommended = "isolation_forest" + + artifacts: List[Dict[str, str]] = [ + {"name": "anomaly_toolkit.py", "language": "python", "content": ANOMALY_CODE}, + {"name": "MODEL_SELECTION.md", "language": "markdown", "content": MODEL_SELECTION}, + ] + + return SkillResult( + success=True, + output=( + f"[anomaly_detection] recommended={recommended}\n" + f"Generated toolkit with 5 detectors + PCA viz + evaluation harness." + ), + artifacts=artifacts, + suggestions=[ + "Plot score distribution to choose contamination threshold", + "Always scale features (StandardScaler / RobustScaler) before fitting", + "Combine unsupervised scores with rule-based features for fraud", + "Track precision@k instead of recall when ground-truth is partial", + "Retrain periodically — anomaly patterns drift over time", + ], + metadata={ + "skill": self.name, + "recommended_model": recommended, + "models_available": [ + "isolation_forest", "lof", "one_class_svm", + "dbscan", "z_score", "iqr", + ], + "version": self.version, + "author": self.author, + }, + ) diff --git a/nexus/skills/api_design.py b/nexus/skills/api_design.py new file mode 100644 index 0000000000000000000000000000000000000000..2951ff1f72e3433b9421628f72a1edae1e3daaf7 --- /dev/null +++ b/nexus/skills/api_design.py @@ -0,0 +1,416 @@ +"""API Design Skill - Sinh OpenAPI 3.0 spec + REST API templates. + +Cung cấp template cho RESTful API design: resource naming, status codes, +pagination, versioning, error format, idempotency, và OpenAPI 3.0 spec. + +Author: Hieu Louis (2026) +""" +from __future__ import annotations + +from typing import Dict, List + +from .base import Skill, SkillContext, SkillCategory, SkillPriority, SkillResult + + +class APIDesignSkill(Skill): + """Sinh REST API design + OpenAPI 3.0 spec template.""" + + category = SkillCategory.SYSTEM + priority = SkillPriority.HIGH + keywords: List[str] = [ + "api design", "rest api", "restful", "openapi", + "swagger", "api spec", "api documentation", + "endpoint design", "thiết kế api", "resource naming", + ] + examples = [ + "Design a REST API for a blog platform", + "Generate OpenAPI 3.0 spec for users and posts endpoints", + "REST API versioning strategy for breaking changes", + ] + + @property + def name(self) -> str: + return "api_design" + + @property + def description(self) -> str: + return ( + "Sinh REST API design + OpenAPI 3.0 spec: resource naming, " + "status codes, pagination, versioning, errors, idempotency." + ) + + def can_handle(self, prompt: str, context: SkillContext = None) -> float: + prompt_lower = prompt.lower() + score = 0.0 + for kw in self.keywords: + if kw in prompt_lower: + score += 0.16 + return min(1.0, score) + + def execute(self, context: SkillContext) -> SkillResult: + return SkillResult( + success=True, + output="[APIDesign] OpenAPI 3.0 spec template + REST guidelines ready.", + artifacts=[ + {"path": "api/openapi.yaml", "content": _OPENAPI_SPEC}, + {"path": "api/guidelines.md", "content": _REST_GUIDELINES}, + ], + metadata={ + "skill": self.name, + "rest_principles": { + "resource_naming": "Plural nouns, lowercase, hyphenated: /users, /order-items", + "http_methods": { + "GET": "Read (idempotent, cacheable)", + "POST": "Create (non-idempotent — use Idempotency-Key for safety)", + "PUT": "Full replace (idempotent)", + "PATCH": "Partial update (idempotent if operation-based, not if value-based)", + "DELETE": "Remove (idempotent)", + }, + "status_codes": { + "2xx_success": ["200 OK", "201 Created", "202 Accepted", "204 No Content"], + "3xx_redirect": ["301 Moved Permanently", "304 Not Modified"], + "4xx_client_error": ["400 Bad Request", "401 Unauthorized", "403 Forbidden", + "404 Not Found", "409 Conflict", "422 Unprocessable Entity", + "429 Too Many Requests"], + "5xx_server_error": ["500 Internal Server Error", "502 Bad Gateway", + "503 Service Unavailable", "504 Gateway Timeout"], + }, + "versioning": { + "uri_versioning": "/v1/users (simple, visible, cacheable)", + "header_versioning": "Accept: application/vnd.api+json;version=1", + "query_param": "?api-version=1 (rare)", + "media_type": "application/vnd.example.v1+json (most RESTful)", + }, + }, + "pagination": { + "offset_limit": "?page=2&limit=20 — simple, slow on large offsets", + "cursor": "?cursor=abc123 — stable, fast for infinite scroll", + "keyset": "?after_id=1234 — fastest for ordered data", + "links": "Use Link header (RFC 5988) or response body _links", + }, + "errors": { + "format": "RFC 9457 (formerly RFC 7807) Problem Details for HTTP APIs", + "fields": ["type", "title", "status", "detail", "instance", "errors[]"], + "example": "application/problem+json", + }, + "idempotency": { + "header": "Idempotency-Key: ", + "ttl": "24-48 hours", + "scope": "per-client (use API key + key as dedup index)", + "applies_to": "POST, PATCH (write operations); not needed for GET/PUT/DELETE", + }, + "security": { + "auth": ["OAuth 2.0 (PKCE for SPAs)", "API key (server-to-server)", + "JWT (short-lived, refresh tokens)"], + "transport": "TLS 1.3 mandatory; HSTS header", + "rate_limit": "X-RateLimit-Remaining / -Limit / -Reset headers", + }, + "tooling": { + "spec": "OpenAPI 3.1 (latest) — Swagger 2.0 deprecated", + "validators": ["openapi-generator", "oas-validator", "redocly-cli"], + "doc_ui": ["Swagger UI", "Redoc", "Elements (Stoplight)"], + "mock": ["Prism", "WireMock", "MSW"], + "lint": ["Spectral (Stoplight)", "vacuum (daveshanley)"], + }, + }, + suggestions=[ + "Specify resource names (e.g. users, orders, posts)", + "Indicate auth method (OAuth2 / API key / JWT)", + "Mention if pagination cursor or offset preferred", + "Ask for SDK generation (openapi-generator for many languages)", + ], + ) + + +_OPENAPI_SPEC = '''openapi: 3.1.0 +info: + title: Example Blog API + version: 1.0.0 + description: | + RESTful API for a blog platform with users, posts, and comments. + Author: Hieu Louis (2026) + contact: + name: API Support + email: api@example.com + license: + name: MIT + url: https://opensource.org/license/mit + +servers: + - url: https://api.example.com/v1 + description: Production + - url: https://staging-api.example.com/v1 + description: Staging + +security: + - bearerAuth: [] + +tags: + - name: users + description: User account management + - name: posts + description: Blog post CRUD + - name: comments + description: Comments on posts + +paths: + /users: + get: + tags: [users] + summary: List users + operationId: listUsers + parameters: + - $ref: '#/components/parameters/PageParam' + - $ref: '#/components/parameters/LimitParam' + - name: sort + in: query + schema: { type: string, enum: [created_at, -created_at, name] } + responses: + '200': + description: A page of users + headers: + X-Total-Count: + schema: { type: integer } + content: + application/json: + schema: + type: object + required: [data, meta] + properties: + data: + type: array + items: { $ref: '#/components/schemas/User' } + meta: + $ref: '#/components/schemas/PageMeta' + post: + tags: [users] + summary: Create user + operationId: createUser + parameters: + - name: Idempotency-Key + in: header + required: true + schema: { type: string, format: uuid } + requestBody: + required: true + content: + application/json: + schema: { $ref: '#/components/schemas/UserCreate' } + responses: + '201': + description: User created + content: + application/json: + schema: { $ref: '#/components/schemas/User' } + '409': + $ref: '#/components/responses/Conflict' + '422': + $ref: '#/components/responses/Unprocessable' + + /users/{userId}: + parameters: + - $ref: '#/components/parameters/UserIdParam' + get: + tags: [users] + summary: Get user by id + operationId: getUser + responses: + '200': + description: A user + content: + application/json: + schema: { $ref: '#/components/schemas/User' } + '404': + $ref: '#/components/responses/NotFound' + patch: + tags: [users] + summary: Update user + operationId: updateUser + requestBody: + required: true + content: + application/json: + schema: { $ref: '#/components/schemas/UserUpdate' } + responses: + '200': + description: Updated user + content: + application/json: + schema: { $ref: '#/components/schemas/User' } + '404': + $ref: '#/components/responses/NotFound' + delete: + tags: [users] + summary: Delete user + operationId: deleteUser + responses: + '204': { description: Deleted } + +components: + securitySchemes: + bearerAuth: + type: http + scheme: bearer + bearerFormat: JWT + + parameters: + PageParam: + name: page + in: query + schema: { type: integer, minimum: 1, default: 1 } + LimitParam: + name: limit + in: query + schema: { type: integer, minimum: 1, maximum: 100, default: 20 } + UserIdParam: + name: userId + in: path + required: true + schema: { type: string, format: uuid } + + schemas: + User: + type: object + required: [id, email, created_at] + properties: + id: { type: string, format: uuid } + email: { type: string, format: email } + name: { type: string, minLength: 1, maxLength: 100 } + created_at: { type: string, format: date-time } + updated_at: { type: string, format: date-time } + UserCreate: + type: object + required: [email] + properties: + email: { type: string, format: email } + name: { type: string, minLength: 1, maxLength: 100 } + UserUpdate: + type: object + properties: + name: { type: string, minLength: 1, maxLength: 100 } + PageMeta: + type: object + required: [page, limit, total] + properties: + page: { type: integer } + limit: { type: integer } + total: { type: integer } + has_next: { type: boolean } + Error: + type: object + required: [type, title, status] + properties: + type: { type: string, format: uri } + title: { type: string } + status: { type: integer } + detail: { type: string } + instance: { type: string } + errors: + type: array + items: + type: object + properties: + field: { type: string } + message: { type: string } + + responses: + NotFound: + description: Resource not found + content: + application/problem+json: + schema: { $ref: '#/components/schemas/Error' } + Conflict: + description: Conflict with current state + content: + application/problem+json: + schema: { $ref: '#/components/schemas/Error' } + Unprocessable: + description: Validation failed + content: + application/problem+json: + schema: { $ref: '#/components/schemas/Error' } +''' + + +_REST_GUIDELINES = """# REST API Design Guidelines + +## 1. Resource Naming +- Use **plural nouns**: `/users`, `/orders`, `/order-items` (not `/orderItem`). +- Use **hyphens** for multi-word: `/order-items` (not `/order_items` or `/orderitems`). +- Nest for sub-resources: `/users/{userId}/posts`. +- Never use verbs in path: `/users/{id}/posts` not `/users/{id}/getPosts`. + +## 2. HTTP Methods (CRUD mapping) +| Action | Method | Path | Success Codes | Idempotent | +|---------|--------|------------------|---------------------|------------| +| List | GET | /users | 200 | Yes | +| Create | POST | /users | 201 + Location hdr | No* | +| Read | GET | /users/{id} | 200 / 404 | Yes | +| Replace | PUT | /users/{id} | 200 | Yes | +| Update | PATCH | /users/{id} | 200 | No* | +| Delete | DELETE | /users/{id} | 204 / 404 | Yes | + +*Use `Idempotency-Key` header for safe retries on POST/PATCH. + +## 3. Status Codes (most common) +- **200** OK — generic success +- **201** Created — POST success (include `Location: /users/{id}`) +- **204** No Content — successful but empty body (DELETE) +- **400** Bad Request — malformed syntax +- **401** Unauthorized — auth missing/invalid +- **403** Forbidden — authenticated but no permission +- **404** Not Found — resource does not exist +- **409** Conflict — duplicate / state violation +- **422** Unprocessable Entity — semantic validation failure +- **429** Too Many Requests — rate limited +- **500** Internal Server Error — bug +- **503** Service Unavailable — maintenance / overload + +## 4. Pagination +- **offset/limit**: simple but slow on large datasets (O(offset)) +- **cursor**: opaque token, stable, fast (preferred for public APIs) +- **keyset**: `?after_id=1234` — fastest for ordered data +- Always return pagination metadata: `page`, `limit`, `total`, `has_next` + +## 5. Versioning +- **URI versioning** (most common): `/v1/users` — simple, visible, cacheable +- **Media type versioning** (most RESTful): `Accept: application/vnd.example.v1+json` +- **Header versioning**: `Api-Version: 1` — invisible, harder to test +- Avoid query param versioning (breaks caching) + +## 6. Error Format (RFC 9457) +``` +HTTP/1.1 422 Unprocessable Entity +Content-Type: application/problem+json + +{ + "type": "https://example.com/errors/validation", + "title": "Validation failed", + "status": 422, + "detail": "Email already in use", + "instance": "/users", + "errors": [ + { "field": "email", "message": "must be unique" } + ] +} +``` + +## 7. Idempotency +- Header: `Idempotency-Key: ` +- Store: `(api_key, idempotency_key) -> response` with 24-48h TTL +- Same key + same body -> return cached response +- Same key + different body -> 422 Conflict (likely client bug) + +## 8. Security +- TLS 1.3 mandatory; HSTS header +- Auth: OAuth 2.0 PKCE for SPAs, API key for server-to-server, JWT short-lived +- Rate limit: `X-RateLimit-Limit`, `-Remaining`, `-Reset` headers +- Never expose internal errors / stack traces in production +- Audit log all mutations + +## 9. Documentation +- Generate from OpenAPI 3.1 spec (single source of truth) +- Swagger UI / Redoc for interactive docs +- Provide examples in multiple languages (curl, Python, JS) +- Changelog with breaking vs non-breaking tags +""" diff --git a/nexus/skills/base.py b/nexus/skills/base.py new file mode 100644 index 0000000000000000000000000000000000000000..306d5ab5a89598519e7b3a7d7a954b64aae2866e --- /dev/null +++ b/nexus/skills/base.py @@ -0,0 +1,144 @@ +""" +Skill Base Class - Nền tảng cho tất cả skills +============================================== +Định nghĩa interface chung cho mọi skill trong Nexus Coder. +""" +from __future__ import annotations + +from abc import ABC, abstractmethod +from dataclasses import dataclass, field +from typing import Any, Dict, List, Optional +from enum import Enum + + +class SkillCategory(str, Enum): + """Phân loại skills theo domain.""" + CODE = "code" + REASONING = "reasoning" + LANGUAGE = "language" + DATA = "data" + DEVOPS = "devops" + SECURITY = "security" + # v0.3 NEW categories + ML = "ml" + CLOUD = "cloud" + SYSTEM = "system" + BLOCKCHAIN = "blockchain" + DATABASE = "database" + NETWORK = "network" + ALGORITHM = "algorithm" + DOCUMENTATION = "documentation" + TESTING = "testing" + MATH = "math" + + +class SkillPriority(str, Enum): + """Độ ưu tiên khi nhiều skills match.""" + LOW = "low" + MEDIUM = "medium" + HIGH = "high" + CRITICAL = "critical" + + +@dataclass +class SkillContext: + """Context passed to skill khi execute. + + Attributes: + prompt: Câu lệnh từ user + language: Ngôn ngữ lập trình (nếu có) + files: Danh sách file liên quan + history: Lịch sử hội thoại + metadata: Extra metadata + max_tokens: Giới hạn output + temperature: Sampling temperature + """ + prompt: str = "" + language: Optional[str] = None + files: List[str] = field(default_factory=list) + history: List[Dict[str, str]] = field(default_factory=list) + metadata: Dict[str, Any] = field(default_factory=dict) + max_tokens: int = 4096 + temperature: float = 0.7 + + +@dataclass +class SkillResult: + """Kết quả trả về từ skill. + + Attributes: + success: Có thành công không + output: Output text + artifacts: Files/code được tạo + suggestions: Gợi ý tiếp theo + error: Thông báo lỗi nếu có + metadata: Extra metadata + """ + success: bool = True + output: str = "" + artifacts: List[Dict[str, str]] = field(default_factory=list) + suggestions: List[str] = field(default_factory=list) + error: Optional[str] = None + metadata: Dict[str, Any] = field(default_factory=dict) + + +class Skill(ABC): + """Base class cho mọi skill trong Nexus Coder. + + Mỗi skill phải implement: + - name: Tên định danh duy nhất + - description: Mô tả ngắn + - execute: Hàm chính thực thi skill + - can_handle: Kiểm tra xem skill có xử lý được prompt không + """ + + category: SkillCategory = SkillCategory.CODE + priority: SkillPriority = SkillPriority.MEDIUM + keywords: List[str] = [] + examples: List[str] = [] + + @property + @abstractmethod + def name(self) -> str: + """Tên duy nhất của skill (snake_case).""" + ... + + @property + @abstractmethod + def description(self) -> str: + """Mô tả ngắn gọn skill làm gì.""" + ... + + @property + def version(self) -> str: + return "0.3.0" + + @property + def author(self) -> str: + return "Hieu Louis" + + def can_handle(self, prompt: str, context: SkillContext = None) -> float: + """Trả về confidence score [0.0, 1.0] cho prompt này. + + Default implementation: match keywords. + Override để implement logic phức tạp hơn. + """ + if not prompt: + return 0.0 + prompt_lower = prompt.lower() + if not self.keywords: + return 0.1 + matches = sum(1 for kw in self.keywords if kw.lower() in prompt_lower) + return min(1.0, matches / max(1, len(self.keywords)) * 2) + + @abstractmethod + def execute(self, context: SkillContext) -> SkillResult: + """Thực thi skill với context đã cho.""" + ... + + def get_system_prompt(self) -> str: + """System prompt đặc thù cho skill (dùng khi gọi LLM).""" + return f"You are using the {self.name} skill. {self.description}" + + def __repr__(self) -> str: + return f"" diff --git a/nexus/skills/blockchain_audit.py b/nexus/skills/blockchain_audit.py new file mode 100644 index 0000000000000000000000000000000000000000..eb755b4dbff862166cfa6baf102d9bbd309e8708 --- /dev/null +++ b/nexus/skills/blockchain_audit.py @@ -0,0 +1,177 @@ +"""Blockchain Audit Skill - Smart contract audit checklist + vulnerability patterns. + +Hỗ trợ Solidity / EVM chains. Sinh audit checklist, common +vulnerability patterns (reentrancy, overflow, ...), và security +best practices (OpenZeppelin, slither, mythril). + +Author: Hieu Louis (2026) +""" +from __future__ import annotations + +from typing import Dict, List + +from .base import Skill, SkillContext, SkillCategory, SkillPriority, SkillResult + + +class BlockchainAuditSkill(Skill): + """Audit smart contracts: checklist + common vulnerabilities + tooling.""" + + category = SkillCategory.BLOCKCHAIN + priority = SkillPriority.HIGH + keywords: List[str] = [ + "solidity", "smart contract", "audit", "erc20", "erc721", + "erc1155", "web3", "ethereum", "evm", "foundry", "hardhat", + "reentrancy", "overflow", "underflow", "governance", + "defi", "flash loan", "proxy", "upgradeable", + ] + examples = [ + "Audit this ERC20 contract for reentrancy", + "Check governance contract for known vulnerabilities", + "Run slither + mythril on my Solidity code", + ] + + @property + def name(self) -> str: + return "blockchain_audit" + + @property + def description(self) -> str: + return ( + "Audit smart contracts (Solidity / EVM): checklist toàn diện, " + "common vulnerability patterns (reentrancy, integer overflow, " + "access control, oracle manipulation, flash loan attacks), " + "và static analysis tooling (Slither, Mythril, Echidna, Foundry fuzz)." + ) + + def can_handle(self, prompt: str, context: SkillContext = None) -> float: + prompt_lower = prompt.lower() + score = 0.0 + for kw in self.keywords: + if kw in prompt_lower: + score += 0.15 + if any(k in prompt_lower for k in (".sol", "pragma solidity", "contract ", "function ")) and "solidity" in prompt_lower: + score += 0.3 + return min(1.0, score) + + def execute(self, context: SkillContext) -> SkillResult: + return SkillResult( + success=True, + output=f"[BlockchainAudit] {len(_VULNS)} vulnerability patterns + checklist ready.", + artifacts=[{"path": "audit/checklist.md", "content": self._render_checklist()}], + metadata={ + "skill": self.name, + "audit_phases": [ + "1. Manual review (logic, access control)", + "2. Static analysis (Slither, Mythril)", + "3. Fuzz + invariant testing (Echidna, Foundry)", + "4. Formal verification (Certora, Halmos) — optional", + "5. Gas optimization review", + "6. Cross-check against SWC registry", + ], + "vulnerabilities": _VULNS, + "tools": { + "static": ["slither", "mythril", "solhint", "securify"], + "fuzz": ["echidna", "foundry fuzz", "medusa"], + "formal": ["certora", "halmos", "keccak"], + "monitoring": ["forta", "openzeppelin defender"], + }, + "standards": ["SWC Registry", "OpenZeppelin Contracts", "EIP-20/721/1155/4626"], + "severity": ["critical", "high", "medium", "low", "info"], + }, + suggestions=[ + "Use OpenZeppelin's SafeERC20 + ReentrancyGuard — never roll your own", + "Run slither in CI on every PR: slither . --exclude-dependencies", + "Add invariant tests with Foundry (testFuzz_* and invariant_*)", + "Get a third-party audit before mainnet — never self-audit for production", + "Time-lock + multisig on governance (>= 48h timelock)", + ], + ) + + def _render_checklist(self) -> str: + lines = ["# Smart Contract Audit Checklist", ""] + for v in _VULNS: + lines.append(f"## {v['id']}: {v['name']} (severity: {v['severity']})") + lines.append(f"**Description:** {v['description']}") + lines.append(f"**Mitigation:** {v['mitigation']}") + lines.append("") + return "\n".join(lines) + + +_VULNS: List[Dict[str, str]] = [ + { + "id": "SWC-107", + "name": "Reentrancy", + "severity": "critical", + "description": ( + "External call to untrusted contract re-enters the function " + "before state is updated, draining funds." + ), + "mitigation": ( + "Use Checks-Effects-Interactions pattern + ReentrancyGuard. " + "Pull payments over push. Use OpenZeppelin's nonReentrant." + ), + }, + { + "id": "SWC-101", + "name": "Integer Overflow / Underflow", + "severity": "high", + "description": "Arithmetic wraps around (pre-0.8 Solidity).", + "mitigation": "Use Solidity >= 0.8 (built-in overflow checks) or SafeMath.", + }, + { + "id": "SWC-105", + "name": "Unauthorized Access / Missing Access Control", + "severity": "critical", + "description": "Functions callable by anyone (mint, withdraw, pause).", + "mitigation": "Use onlyRole / ownable / access control. Prefer RBAC.", + }, + { + "id": "SWC-116", + "name": "Block Timestamp Manipulation", + "severity": "medium", + "description": "Miners can tweak block.timestamp within ~15s.", + "mitigation": "Don't use block.timestamp for strict randomness or critical logic.", + }, + { + "id": "SWC-114", + "name": "Transaction Order Dependence (Front-running)", + "severity": "high", + "description": "Adversary sees mempool tx and front-runs.", + "mitigation": "Commit-reveal scheme, slippage tolerance, MEV-protected routers.", + }, + { + "id": "SWC-113", + "name": "DoS via Block Gas Limit / Unbounded Loop", + "severity": "high", + "description": "Loop over dynamic array grows past block gas limit -> permanent DoS.", + "mitigation": "Cap iterations; split into batches; avoid storing rewards in growing arrays.", + }, + { + "id": "ORACLE", + "name": "Oracle Manipulation / Flash Loan Attack", + "severity": "critical", + "description": "Single-DEX price oracle spoofable via flash loans.", + "mitigation": "Use TWAP (Uniswap V3), Chainlink aggregators, or median of multiple sources.", + }, + { + "id": "PROXY", + "name": "Upgradeable Proxy Storage Collision", + "severity": "high", + "description": "Logic contract state vars collide with proxy admin slot.", + "mitigation": "Use EIP-1967 transparent / UUPS proxies with storage gaps; OpenZeppelin upgrades plugin.", + }, + { + "id": "GOV", + "name": "Governance / Flash Loan Voting", + "severity": "high", + "description": "Attacker borrows tokens, votes, repays in one tx.", + "mitigation": "Snapshot voting + lock-up periods (e.g., veToken).", + }, + { + "id": "GAS", + "name": "Gas Griefing / Unbounded Refund", + "severity": "medium", + "description": "Recipient contract's fallback blocks ether transfer.", + "mitigation": "Use .call with value + checks-effects-interactions; CEI pattern.", + }, +] diff --git a/nexus/skills/bug_reproduction.py b/nexus/skills/bug_reproduction.py new file mode 100644 index 0000000000000000000000000000000000000000..1dd6cac31702bbca099cf6d4d42caada29849444 --- /dev/null +++ b/nexus/skills/bug_reproduction.py @@ -0,0 +1,174 @@ +"""Bug Reproduction Skill - Minimal Reproducible Example (MRE) framework. + +Sinh khung reproduce bug: isolation steps, environment snapshot, +minimal repro script, và bisect strategy cho Git history. + +Author: Hieu Louis (2026) +""" +from __future__ import annotations + +from typing import Dict, List + +from .base import Skill, SkillCategory, SkillContext, SkillPriority, SkillResult + + +REPRO_TEMPLATE = """ +# Bug Reproduction Report / Báo cáo tái lập bug + +**Bug ID:** {bug_id} +**Title:** {title} +**Severity:** {severity} (blocker / critical / major / minor / trivial) +**Reported:** {reported_at} + +## 1. Environment Snapshot / Môi trường +- OS: `uname -a` +- Runtime: python --version (or node -v / go version) +- Dependencies: + pip freeze > requirements-bug.txt # pin exact versions +- Repo state: + git rev-parse HEAD + git status --short + git log -1 --format='%H %s' + +## 2. Preconditions / Điều kiện tiên quyết +- ... +- ... + +## 3. Steps to Reproduce / Các bước tái lập +1. +2. +3. + +## 4. Expected vs. Actual / Kỳ vọng vs Thực tế +- Expected: +- Actual: + +## 5. Minimal Reproducible Example (MRE) / Ví dụ tối thiểu +- Strip everything unrelated to the bug. +- Hard-code inputs (no DB / network if possible). +- Target ≤ 50 lines. + +## 6. Frequency / Tần suất +- Always | Intermittent (x% of runs) | Only on CI + +## 7. Logs / Traces +- Stack trace, stderr, screenshots, profiler output attached + +## 8. Suspected Root Cause / Nghi ngờ nguyên nhân +- ... + +## 9. Workaround / Tạm thời +- ... +""" + +MRE_PYTHON = '''"""MRE: .""" +from __future__ import annotations +import sys, platform, traceback + +print(f"python={sys.version} | os={platform.platform()}") + +def repro() -> None: + """Reproduce the bug deterministically.""" + # --- Arrange --- minimal setup with hard-coded inputs + data = [1, 2, 3, None, 5] + + # --- Act --- the smallest call that triggers the bug + try: + result = sum(x or 0 for x in data) + except Exception: + traceback.print_exc() + return + + # --- Assert --- what should happen vs what actually happens + expected = 11 + assert result == expected, f"BUG: got {result}, expected {expected}" + print("No bug reproduced — adjust inputs / version.") + +if __name__ == "__main__": + repro() +''' + +BISECT_SCRIPT = '''#!/usr/bin/env bash +# git bisect driver — exit 0=good, 1=bad, 125=skip +# Usage: git bisect start BAD GOOD -- && git bisect run ./bisect.sh +set -euo pipefail +python -m pytest tests/test_repro.py -q || exit 1 +exit 0 +''' + + +class BugReproductionSkill(Skill): + """Tạo khung Minimal Reproducible Example cho bug reports.""" + + category = SkillCategory.CODE + priority = SkillPriority.HIGH + keywords: List[str] = [ + "bug", "reproduce", "repro", "mre", "minimal example", + "minimal reproducible", "regression", "regression test", + "bisect", "stack trace", "traceback", + ] + examples = [ + "Tôi gặp bug X khi chạy Y, giúp tạo repro", + "Reproduce bug từ stack trace này", + "Tạo minimal example cho crash", + ] + + @property + def name(self) -> str: + return "bug_reproduction" + + @property + def description(self) -> str: + return ( + "Sinh khung Minimal Reproducible Example (MRE): isolation steps, " + "environment snapshot, repro script, và git bisect strategy." + ) + + def can_handle(self, prompt: str, context: SkillContext = None) -> float: + prompt_lower = prompt.lower() + score = 0.0 + for kw in self.keywords: + if kw in prompt_lower: + score += 0.22 + return min(1.0, score) + + def execute(self, context: SkillContext) -> SkillResult: + bug_id = context.metadata.get("bug_id", "BUG-001") + title = context.metadata.get("title", (context.prompt or "")[:80]) + severity = context.metadata.get("severity", "major") + reported_at = context.metadata.get("reported_at", "2026-01-15") + + report = REPRO_TEMPLATE.format( + bug_id=bug_id, title=title, severity=severity, reported_at=reported_at + ) + + artifacts: List[Dict[str, str]] = [ + {"name": "BUG_REPORT.md", "language": "markdown", "content": report}, + {"name": "repro.py", "language": "python", "content": MRE_PYTHON}, + {"name": "bisect.sh", "language": "bash", "content": BISECT_SCRIPT}, + ] + + return SkillResult( + success=True, + output=( + f"[bug_reproduction] bug_id={bug_id} severity={severity}\n" + f"Generated MRE framework: report + repro.py + bisect.sh" + ), + artifacts=artifacts, + suggestions=[ + "Reduce the repro script until removing any line stops the bug", + "Add `pytest -p no:randomly` if order-dependent", + "Use `git bisect run ./bisect.sh` to localize the regression commit", + "Attach heap profilers (tracemalloc / memray) for memory bugs", + "If flaky: run 100× with `pytest --repeats 100` to estimate frequency", + ], + metadata={ + "skill": self.name, + "bug_id": bug_id, + "severity": severity, + "title": title, + "has_bisect": True, + "version": self.version, + "author": self.author, + }, + ) diff --git a/nexus/skills/caching_strategy.py b/nexus/skills/caching_strategy.py new file mode 100644 index 0000000000000000000000000000000000000000..3b8bce57dfa16dc42e63f2f2631d307141c9208c --- /dev/null +++ b/nexus/skills/caching_strategy.py @@ -0,0 +1,178 @@ +"""Caching Strategy Skill - Cache design analysis + Redis / Memcached config. + +Phân tích chiến lược cache: write-through, write-back, cache-aside, refresh-ahead; +TTL & eviction policy; stampede protection (lock + jitter); Redis config mẫu. + +Author: Hieu Louis (2026) +""" +from __future__ import annotations + +from typing import Dict, List + +from .base import Skill, SkillCategory, SkillContext, SkillPriority, SkillResult + + +CACHE_STRATEGIES = """ +Cache Strategies Comparison / So sánh chiến lược cache +======================================================= + +| Strategy | Write Path | Pros | Cons | +|-----------------|-------------------|-----------------------------|-------------------------------| +| Cache-Aside | App updates both | Simple, resilient | Stale on failure | +| Write-Through | Cache then DB | Strong consistency | Higher write latency | +| Write-Back | Cache only (async)| Fast writes | Data loss risk on crash | +| Refresh-Ahead | Pre-emptive | Hides latency for hot keys | Extra infra & complexity | + +Eviction Policies: LRU (Redis default), LFU, FIFO, TTL-based +Invalidation: explicit `DEL`, key-bucket versioning, pub/sub bust, tag-based (Redis 7.4) + +Stampede Protection: + - Single-flight lock (SET NX EX 30) before recompute + - Early refresh (TTL * 0.8) with random jitter + - Bloom filter for negative caching (anti cache-penetration) +""" + +REDIS_CONFIG = """# redis.conf — Production tuning / Cấu hình production +bind 0.0.0.0 +protected-mode yes +port 6379 +tcp-keepalive 300 +timeout 0 + +# Memory & eviction / Bộ nhớ & loại bỏ +maxmemory 4gb +maxmemory-policy allkeys-lru # evict least-recently-used +lfu-log-factor 10 + +# Persistence / Độ bền dữ liệu +appendonly yes +appendfsync everysec # balance durability vs throughput +auto-aof-rewrite-percentage 100 +auto-aof-rewrite-min-size 64mb +save 900 1 +save 300 10 + +# Replication / Sao chép +replica-read-only yes +repl-backlog-size 64mb + +# Security / Bảo mật +requirepass ${REDIS_PASSWORD} +rename-command FLUSHDB "" +rename-command FLUSHALL "" +rename-command KEYS "" +""" + +STAMPEDE_GUARD = '''"""Cache-aside with stampede protection / Cache-aside chống dồn dập.""" +import json, time, random, uuid +import redis + +r = redis.Redis(host="redis", port=6379, decode_responses=True) +LOCK_TTL = 30 # seconds +BASE_TTL = 3600 + +def cached(key: str, loader, ttl: int = BASE_TTL): + val = r.get(key) + if val is not None: + return json.loads(val) + + # Single-flight: acquire lock to recompute / Chỉ 1 worker recompute + lock = f"{key}:lock" + token = str(uuid.uuid4()) + if not r.set(lock, token, nx=True, ex=LOCK_TTL): + time.sleep(0.05 + random.random() * 0.1) # jitter + return cached(key, loader, ttl) # retry read + + try: + val = loader() + # Early-refresh: keep value warm but mark as stale soon / Giữ ấm giá trị + effective_ttl = int(ttl * 0.8) + random.randint(0, 60) + r.setex(key, effective_ttl, json.dumps(val)) + return val + finally: + # Lua CAS to safely release our lock (avoid removing others') + r.eval( + "if redis.call('get',KEYS[1])==ARGV[1] then return redis.call('del',KEYS[1]) end", + 1, lock, token, + ) +''' + + +class CachingStrategySkill(Skill): + """Phân tích & sinh caching strategy + Redis/Memcached config.""" + + category = SkillCategory.SYSTEM + priority = SkillPriority.MEDIUM + keywords: List[str] = [ + "cache", "caching", "redis", "memcached", "cache invalidation", + "cache stampede", "cache-aside", "write-through", "eviction", + "ttl", "lru", "lfu", "cdn", + ] + examples = [ + "Thiết kế caching cho API endpoint", + "Setup Redis cluster with stampede protection", + "Choose cache eviction policy for session store", + ] + + @property + def name(self) -> str: + return "caching_strategy" + + @property + def description(self) -> str: + return ( + "Phân tích cache strategy (cache-aside / write-through / write-back) " + "+ sinh Redis config với stampede protection và eviction tuning." + ) + + def can_handle(self, prompt: str, context: SkillContext = None) -> float: + prompt_lower = prompt.lower() + score = 0.0 + for kw in self.keywords: + if kw in prompt_lower: + score += 0.15 + return min(1.0, score) + + def execute(self, context: SkillContext) -> SkillResult: + prompt_lower = (context.prompt or "").lower() + # Pick default strategy / Chọn strategy mặc định + if "write-through" in prompt_lower: + strategy = "write_through" + elif "write-back" in prompt_lower or "writeback" in prompt_lower: + strategy = "write_back" + elif "refresh-ahead" in prompt_lower or "refresh_ahead" in prompt_lower: + strategy = "refresh_ahead" + else: + strategy = "cache_aside" + + backend = "memcached" if "memcached" in prompt_lower else "redis" + + artifacts: List[Dict[str, str]] = [ + {"name": "CACHE_STRATEGIES.md", "language": "markdown", "content": CACHE_STRATEGIES}, + {"name": "redis.conf", "language": "ini", "content": REDIS_CONFIG}, + {"name": "stampede_guard.py", "language": "python", "content": STAMPEDE_GUARD}, + ] + + return SkillResult( + success=True, + output=( + f"[caching_strategy] strategy={strategy} | backend={backend}\n" + f"Generated strategy doc + {backend} config + stampede guard." + ), + artifacts=artifacts, + suggestions=[ + "Benchmark with realistic read/write ratio (e.g. 95/5 read-heavy)", + "Add monitoring: hit ratio, latency p99, eviction rate, memory usage", + "Use Redis Sentinel / Cluster for HA in production", + "Negative cache empty results to prevent cache-penetration", + "Tag-based invalidation (Redis 7.4+) for multi-key busts", + ], + metadata={ + "skill": self.name, + "strategy": strategy, + "backend": backend, + "eviction_policy": "allkeys-lru", + "version": self.version, + "author": self.author, + }, + ) diff --git a/nexus/skills/ci_cd_pipeline.py b/nexus/skills/ci_cd_pipeline.py new file mode 100644 index 0000000000000000000000000000000000000000..3141424d82a50b5c65214f52e63df556cde770ab --- /dev/null +++ b/nexus/skills/ci_cd_pipeline.py @@ -0,0 +1,238 @@ +"""CI/CD Pipeline Skill - Sinh pipeline YAML cho GitHub Actions / GitLab CI / Jenkins. + +Cung cấp template pipeline CI/CD hoàn chỉnh: build, test, scan, publish, +deploy với chiến lược branch (trunk-based / GitFlow) và environment promotion. + +Author: Hieu Louis (2026) +""" +from __future__ import annotations + +from typing import Dict, List + +from .base import Skill, SkillCategory, SkillContext, SkillPriority, SkillResult + + +# Template: GitHub Actions / GitLab CI / Jenkins +GITHUB_ACTIONS_TEMPLATE = """# .github/workflows/ci.yml (GitHub Actions) +name: CI + +on: + push: + branches: [main, develop] + pull_request: + branches: [main] + +permissions: + contents: read + packages: write + +jobs: + build-test: + runs-on: ubuntu-latest + timeout-minutes: 30 + steps: + - uses: actions/checkout@v4 + with: + fetch-depth: 0 # cần cho cache key & changelog + + - name: Setup Python + uses: actions/setup-python@v5 + with: + python-version: "3.12" + cache: "pip" + + - name: Install deps + run: | + python -m pip install --upgrade pip + pip install -r requirements.txt + pip install ruff pytest pytest-cov safety bandit + + - name: Lint (ruff) + run: ruff check . + + - name: Type-check (mypy) + run: mypy --strict nexus + + - name: Test + coverage + run: pytest --cov=nexus --cov-report=xml --cov-report=term-missing + + - name: SAST (bandit) + dependency scan (safety) + run: | + bandit -r nexus -q + safety check --short + + - name: Build artifacts + run: python -m build + + - name: Upload coverage + if: github.event_name == 'push' + uses: codecov/codecov-action@v4 + + publish: + needs: build-test + if: github.ref == 'refs/heads/main' + runs-on: ubuntu-latest + environment: production + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-python@v5 + with: { python-version: "3.12", cache: "pip" } + - run: pip install build twine + - run: python -m build + - run: twine upload dist/* + env: + TWINE_API_TOKEN: ${{ secrets.PYPI_TOKEN }} +""" + +GITLAB_CI_TEMPLATE = """# .gitlab-ci.yml (GitLab CI) +stages: [lint, test, build, deploy] + +variables: + PIP_CACHE_DIR: "$CI_PROJECT_DIR/.cache/pip" + PYTHON_IMAGE: "python:3.12-slim" + +cache: + key: "$CI_COMMIT_REF_SLUG" + paths: [.cache/pip, .venv/] + +lint: + stage: lint + image: $PYTHON_IMAGE + script: + - pip install ruff mypy + - ruff check . + - mypy --strict nexus + +test: + stage: test + image: $PYTHON_IMAGE + script: + - pip install -r requirements.txt pytest pytest-cov + - pytest --cov=nexus --cov-report=xml + artifacts: + reports: + coverage_report: + coverage_format: cobertura + path: coverage.xml + coverage: '/TOTAL.*\\s+(\\d+\\%)$/' + +build: + stage: build + image: $PYTHON_IMAGE + script: python -m build + artifacts: + paths: [dist/] + rules: + - if: $CI_COMMIT_TAG + +deploy:prod: + stage: deploy + image: $PYTHON_IMAGE + environment: production + script: + - pip install twine + - twine upload dist/* + rules: + - if: $CI_COMMIT_TAG =~ /^v\\d+\\.\\d+\\.\\d+$/ + when: manual +""" + + +class CICDPipelineSkill(Skill): + """Sinh CI/CD pipeline template cho GitHub Actions / GitLab CI / Jenkins.""" + + category = SkillCategory.DEVOPS + priority = SkillPriority.HIGH + keywords: List[str] = [ + "ci/cd", "ci cd", "cicd", "pipeline", "jenkins", + "github actions", "gitlab ci", "gitlab-ci", "continuous integration", + "continuous deployment", "workflow", "ci build", + ] + examples = [ + "Tạo CI/CD pipeline cho Python project dùng GitHub Actions", + "Setup GitLab CI với test + build + deploy", + "Configure Jenkins pipeline cho microservice", + ] + + @property + def name(self) -> str: + return "ci_cd_pipeline" + + @property + def description(self) -> str: + return ( + "Sinh CI/CD pipeline templates (GitHub Actions / GitLab CI / Jenkins) " + "với lint, test, SAST, build, publish và environment-gated deploy." + ) + + def can_handle(self, prompt: str, context: SkillContext = None) -> float: + prompt_lower = prompt.lower() + score = 0.0 + for kw in self.keywords: + if kw in prompt_lower: + score += 0.22 + return min(1.0, score) + + def execute(self, context: SkillContext) -> SkillResult: + # Chọn engine dựa trên prompt / Chọn engine theo keyword + prompt_lower = (context.prompt or "").lower() + if "gitlab" in prompt_lower: + engine, template, filename = "gitlab_ci", GITLAB_CI_TEMPLATE, ".gitlab-ci.yml" + elif "jenkins" in prompt_lower: + engine, template, filename = "jenkins", ( + "# Jenkinsfile (Declarative)\n" + "pipeline {\n" + " agent any\n" + " options { timeout(time: 30, unit: 'MINUTES') }\n" + " stages {\n" + " stage('Lint') { steps { sh 'ruff check .' } }\n" + " stage('Test') { steps { sh 'pytest --cov=nexus' } }\n" + " stage('Build') { steps { sh 'python -m build' } }\n" + " stage('Deploy') {\n" + " when { branch 'main' }\n" + " steps { sh 'twine upload dist/*' }\n" + " }\n" + " }\n" + "}\n" + ), "Jenkinsfile" + else: + engine, template, filename = "github_actions", GITHUB_ACTIONS_TEMPLATE, ".github/workflows/ci.yml" + + stages = ["lint", "test", "sast", "build", "publish", "deploy"] + artifacts: List[Dict[str, str]] = [ + {"name": filename, "language": "yaml", "content": template}, + { + "name": "BRANCHING.md", + "language": "markdown", + "content": ( + "# Branch Strategy / Chiến lược nhánh\n\n" + "- `main` : luôn deployable (trunk-based)\n" + "- `develop` : integration branch (GitFlow optional)\n" + "- `feat/*` : short-lived feature branches\n" + "- Tag `vMAJOR.MINOR.PATCH` → trigger release\n" + ), + }, + ] + + return SkillResult( + success=True, + output=( + f"[ci_cd_pipeline] engine={engine} | stages={','.join(stages)}\n" + f"Generated {filename} ({len(template)} bytes) with branch strategy guide." + ), + artifacts=artifacts, + suggestions=[ + "Add matrix build for multiple Python versions if cross-version support is required", + "Enable required status checks + branch protection on main", + "Configure environment secrets per stage (staging → production)", + "Add a dependabot/renovate workflow to keep actions pinned", + ], + metadata={ + "skill": self.name, + "engine": engine, + "stages": stages, + "filename": filename, + "version": self.version, + "author": self.author, + }, + ) diff --git a/nexus/skills/classification_automation.py b/nexus/skills/classification_automation.py new file mode 100644 index 0000000000000000000000000000000000000000..8987c9fdb9cba3d93c85d47da2e5e7fe66c885bd --- /dev/null +++ b/nexus/skills/classification_automation.py @@ -0,0 +1,262 @@ +"""Classification Automation Skill - sklearn classification pipeline. + +Sinh end-to-end classification pipeline: data splitting (stratified), +preprocessing (numeric + categorical transformers), model zoo +(LogisticRegression / RandomForest / XGBoost / LightGBM / SVM / KNN), +cross-validation, hyperparameter search, evaluation (metrics + ROC/PR curves), +và feature importance. Handles class imbalance (SMOTE / class_weight). + +Author: Hieu Louis (2026) +""" +from __future__ import annotations + +from typing import Dict, List + +from .base import Skill, SkillCategory, SkillContext, SkillPriority, SkillResult + + +CLASSIFICATION_PIPELINE = '''"""End-to-end classification pipeline / Pipeline phân loại full.""" +from __future__ import annotations +from typing import Dict, Tuple +import numpy as np +import pandas as pd +from sklearn.model_selection import ( + train_test_split, StratifiedKFold, cross_validate, GridSearchCV, +) +from sklearn.compose import ColumnTransformer +from sklearn.pipeline import Pipeline +from sklearn.preprocessing import StandardScaler, OneHotEncoder +from sklearn.impute import SimpleImputer +from sklearn.linear_model import LogisticRegression +from sklearn.ensemble import RandomForestClassifier, GradientBoostingClassifier +from sklearn.svm import SVC +from sklearn.neighbors import KNeighborsClassifier +from sklearn.metrics import ( + accuracy_score, precision_score, recall_score, f1_score, roc_auc_score, + classification_report, confusion_matrix, +) +try: + from xgboost import XGBClassifier + from lightgbm import LGBMClassifier + from imblearn.over_sampling import SMOTE + from imblearn.pipeline import Pipeline as ImbPipeline + HAS_XGB = HAS_LGB = HAS_IMB = True +except ImportError: + HAS_XGB = HAS_LGB = HAS_IMB = False + + +def build_preprocessor(df: pd.DataFrame, target: str) -> ColumnTransformer: + """Tạo ColumnTransformer cho numeric + categorical.""" + numeric = [c for c in df.columns if c != target + and pd.api.types.is_numeric_dtype(df[c])] + categorical = [c for c in df.columns if c != target + and not pd.api.types.is_numeric_dtype(df[c])] + + num_pipe = Pipeline([ + ("imputer", SimpleImputer(strategy="median")), + ("scaler", StandardScaler()), + ]) + cat_pipe = Pipeline([ + ("imputer", SimpleImputer(strategy="most_frequent")), + ("encoder", OneHotEncoder(handle_unknown="ignore", sparse_output=False)), + ]) + return ColumnTransformer([ + ("num", num_pipe, numeric), + ("cat", cat_pipe, categorical), + ], remainder="drop") + + +def build_model_zoo(random_state: int = 42) -> Dict[str, object]: + """Dictionary of candidate classifiers / Bộ mô hình ứng viên.""" + zoo = { + "logreg": LogisticRegression(max_iter=1000, class_weight="balanced", + random_state=random_state), + "random_forest": RandomForestClassifier(n_estimators=300, n_jobs=-1, + class_weight="balanced", + random_state=random_state), + "gbm": GradientBoostingClassifier(random_state=random_state), + "svm_rbf": SVC(kernel="rbf", probability=True, class_weight="balanced", + random_state=random_state), + "knn": KNeighborsClassifier(n_neighbors=5, n_jobs=-1), + } + if HAS_XGB: + zoo["xgboost"] = XGBClassifier( + n_estimators=400, learning_rate=0.05, max_depth=6, + subsample=0.9, colsample_bytree=0.9, n_jobs=-1, + eval_metric="logloss", random_state=random_state, + use_label_encoder=False, + ) + if HAS_LGB: + zoo["lightgbm"] = LGBMClassifier( + n_estimators=400, learning_rate=0.05, num_leaves=63, + subsample=0.9, colsample_bytree=0.9, n_jobs=-1, + class_weight="balanced", random_state=random_state, verbosity=-1, + ) + return zoo + + +def evaluate_models( + df: pd.DataFrame, target: str, cv: int = 5, test_size: float = 0.2, +) -> Tuple[Dict[str, dict], object]: + """Train + cross-validate all models, trả về metrics + best pipeline.""" + X = df.drop(columns=[target]) + y = df[target] + X_tr, X_te, y_tr, y_te = train_test_split( + X, y, test_size=test_size, stratify=y, random_state=42, + ) + + pre = build_preprocessor(df, target) + skf = StratifiedKFold(n_splits=cv, shuffle=True, random_state=42) + results: Dict[str, dict] = {} + best_f1, best_name, best_pipe = -1.0, None, None + + for name, model in build_model_zoo().items(): + steps = [("pre", pre), ("clf", model)] + if HAS_IMB: + pipe = ImbPipeline(steps + [("smote", SMOTE(random_state=42))] if False else steps) + else: + pipe = Pipeline(steps) + try: + cv_res = cross_validate( + pipe, X_tr, y_tr, cv=skf, scoring=["f1_weighted", "roc_auc_ovr_weighted"], + n_jobs=-1, return_train_score=False, error_score="raise", + ) + pipe.fit(X_tr, y_tr) + y_pred = pipe.predict(X_te) + results[name] = { + "cv_f1": float(np.mean(cv_res["test_f1_weighted"])), + "cv_auc": float(np.mean(cv_res["test_roc_auc_ovr_weighted"])), + "test_accuracy": float(accuracy_score(y_te, y_pred)), + "test_f1": float(f1_score(y_te, y_pred, average="weighted")), + "test_precision": float(precision_score(y_te, y_pred, average="weighted", zero_division=0)), + "test_recall": float(recall_score(y_te, y_pred, average="weighted", zero_division=0)), + "report": classification_report(y_te, y_pred, output_dict=True), + } + if results[name]["test_f1"] > best_f1: + best_f1 = results[name]["test_f1"] + best_name, best_pipe = name, pipe + except Exception as e: + results[name] = {"error": str(e)} + return results, (best_name, best_pipe) + + +def grid_search_rf(X, y, pre) -> dict: + """Grid search RandomForest hyper-params / Tối ưu siêu tham số.""" + pipe = Pipeline([("pre", pre), ("clf", RandomForestClassifier(class_weight="balanced", n_jobs=-1))]) + grid = { + "clf__n_estimators": [200, 400, 600], + "clf__max_depth": [None, 10, 20], + "clf__min_samples_leaf": [1, 2, 4], + } + gs = GridSearchCV(pipe, grid, cv=StratifiedKFold(5, shuffle=True, random_state=42), + scoring="f1_weighted", n_jobs=-1, verbose=1) + gs.fit(X, y) + return {"best_params": gs.best_params_, "best_score": float(gs.best_score_)} + + +if __name__ == "__main__": + from sklearn.datasets import load_iris + iris = load_iris(as_frame=True) + df = iris.frame.rename(columns={"target": "y"}) + res, best = evaluate_models(df, "y", cv=5) + print(pd.DataFrame(res).T) + print("best:", best[0]) +''' + +BEST_PRACTICES = """ +Classification Best Practices / Thực hành tốt khi phân loại +============================================================ +1. Splits: + - Stratified train/test (keep class proportions). + - If small data: StratifiedKFold CV (k=5 or 10). + - For time-series: temporal split (no shuffle). + +2. Imbalanced classes: + - class_weight="balanced" in sklearn estimators. + - SMOTE / ADASYN over-sampling (use imblearn Pipeline to avoid leakage). + - Use AUC-PR (not AUC-ROC) + precision@k for rare positives. + - Threshold tuning: optimize F1 / F-beta / cost-based metric on validation. + +3. Leakage prevention: + - Fit preprocessing ONLY on train fold inside Pipeline. + - Do not scale then split — split then scale inside pipeline. + +4. Model selection: + - Start simple (LogReg baseline) → tree ensembles → boosted (XGB/LGBM). + - Compare via CV mean ± std; pick most stable if performance tied. + +5. Calibration: + - For probability-sensitive tasks, apply CalibratedClassifierCV (isotonic). + +6. Hyperparameter search: + - For large search space, prefer Optuna (TPE) over GridSearchCV. +""" + + +class ClassificationSkill(Skill): + """Sinh sklearn classification pipeline (LogReg/RF/XGB/LGBM/SVM/KNN).""" + + category = SkillCategory.ML + priority = SkillPriority.MEDIUM + keywords: List[str] = [ + "classification", "classifier", "logistic regression", + "random forest", "xgboost", "lightgbm", "svm", "knn", + "binary classification", "multiclass", "imbalanced", + ] + examples = [ + "Build classification pipeline cho churn dataset", + "Compare logistic regression vs random forest", + "Handle imbalanced classes with SMOTE", + ] + + @property + def name(self) -> str: + return "classification_automation" + + @property + def description(self) -> str: + return ( + "Sinh end-to-end classification pipeline: preprocessing, model zoo " + "(LogReg/RF/XGB/LGBM/SVM/KNN), stratified CV, grid search, " + "imbalance handling (SMOTE/class_weight) + best-practices guide." + ) + + def can_handle(self, prompt: str, context: SkillContext = None) -> float: + prompt_lower = prompt.lower() + score = 0.0 + for kw in self.keywords: + if kw in prompt_lower: + score += 0.13 + return min(1.0, score) + + def execute(self, context: SkillContext) -> SkillResult: + artifacts: List[Dict[str, str]] = [ + {"name": "classification_pipeline.py", "language": "python", "content": CLASSIFICATION_PIPELINE}, + {"name": "BEST_PRACTICES.md", "language": "markdown", "content": BEST_PRACTICES}, + ] + + return SkillResult( + success=True, + output=( + "[classification_automation] Generated full pipeline: preprocessing " + "(numeric + categorical), 7-model zoo (LogReg/RF/GBM/SVM/KNN/XGB/LGBM), " + "stratified CV + grid search + SMOTE imbalance handling." + ), + artifacts=artifacts, + suggestions=[ + "Start with a LogisticRegression baseline, then try tree ensembles", + "For imbalanced data, optimize threshold on validation F1/PR curve", + "Use CalibratedClassifierCV(isotonic) if probabilities matter", + "Prefer Optuna over GridSearchCV for large hyperparameter spaces", + "Apply SHAP for post-hoc interpretability on the best model", + ], + metadata={ + "skill": self.name, + "models": ["logreg", "random_forest", "gbm", "svm_rbf", "knn", + "xgboost", "lightgbm"], + "handles_imbalance": True, + "metrics": ["accuracy", "precision", "recall", "f1", "roc_auc"], + "version": self.version, + "author": self.author, + }, + ) diff --git a/nexus/skills/cloud_deploy.py b/nexus/skills/cloud_deploy.py new file mode 100644 index 0000000000000000000000000000000000000000..51c58992e50de4eeef68e9379f180ac175ced284 --- /dev/null +++ b/nexus/skills/cloud_deploy.py @@ -0,0 +1,379 @@ +"""Cloud Deploy Skill - Sinh cloud deployment templates. + +Hỗ trợ AWS (EC2/S3/Lambda), GCP (Cloud Run / Compute), Azure +(Container Apps / Functions). Tạo IaC + deploy command. + +Author: Hieu Louis (2026) +""" +from __future__ import annotations + +from typing import Dict, List + +from .base import Skill, SkillContext, SkillCategory, SkillPriority, SkillResult + + +class CloudDeploySkill(Skill): + """Sinh IaC + deploy command cho AWS / GCP / Azure.""" + + category = SkillCategory.CLOUD + priority = SkillPriority.HIGH + keywords: List[str] = [ + "deploy", "deployment", "aws", "gcp", "azure", + "ec2", "s3", "lambda", "cloud run", "cloudrun", + "cloud functions", "ecs", "eks", "fargate", + "compute engine", "container apps", "app service", + "deploy command", "iac", "pulumi", "cdk", + ] + examples = [ + "Deploy FastAPI to AWS Lambda", + "Deploy container to GCP Cloud Run", + "Provision S3 bucket + CloudFront for static site", + ] + + @property + def name(self) -> str: + return "cloud_deploy" + + @property + def description(self) -> str: + return ( + "Sinh cloud deployment templates cho AWS / GCP / Azure: " + "Lambda (SAM), Cloud Run, ECS/Fargate, Container Apps, " + "S3+CloudFront static hosting, với IaC (Terraform / CDK / Pulumi) " + "và deploy commands (aws / gcloud / az)." + ) + + def can_handle(self, prompt: str, context: SkillContext = None) -> float: + prompt_lower = prompt.lower() + score = 0.0 + for kw in self.keywords: + if kw in prompt_lower: + score += 0.15 + return min(1.0, score) + + def _detect_target(self, prompt: str) -> str: + p = prompt.lower() + if "lambda" in p: + return "aws_lambda" + if "cloud run" in p or "cloudrun" in p: + return "gcp_cloudrun" + if "ecs" in p or "fargate" in p: + return "aws_ecs" + if "container apps" in p: + return "azure_containerapps" + if "app service" in p: + return "azure_appservice" + if "compute engine" in p or "gce" in p: + return "gcp_gce" + if "ec2" in p: + return "aws_ec2" + if "s3" in p and ("static" in p or "cloudfront" in p or "website" in p): + return "aws_s3_static" + return "gcp_cloudrun" # sane default for container + + def execute(self, context: SkillContext) -> SkillResult: + target = self._detect_target(context.prompt) + artifact, deploy_cmd = self._build(target) + + return SkillResult( + success=True, + output=f"[CloudDeploy/{target}] IaC + deploy command ready.", + artifacts=[artifact], + metadata={ + "skill": self.name, + "target": target, + "deploy_command": deploy_cmd, + "providers": { + "aws": ["aws-cli", "sam", "terraform", "cdk"], + "gcp": ["gcloud", "terraform", "pulumi"], + "azure": ["az", "terraform", "bicep"], + }, + "checklist": [ + "Pin runtime versions (Python 3.12, Node 20, ...)", + "Set least-privilege IAM role", + "Enable VPC flow logs / CloudTrail / Audit Logs", + "Configure autoscaling + health checks", + "Set up alarms on error rate / latency / cost", + "Store secrets in Secrets Manager / Secret Manager", + ], + }, + suggestions=[ + "Run `terraform plan` in CI before `apply`", + "Use blue/green or canary for production deploys", + "Tag resources with owner / cost-center / env", + "Enable WAF + rate-limiting on public endpoints", + ], + ) + + def _build(self, target: str) -> tuple[Dict[str, str], str]: + if target == "aws_lambda": + return ({"path": "deploy/aws_lambda/template.yaml", "content": _AWS_LAMBDA_SAM}, + "sam build && sam deploy --guided") + if target == "aws_ecs": + return ({"path": "deploy/aws_ecs/main.tf", "content": _AWS_ECS_TF}, + "terraform apply -auto-approve") + if target == "aws_ec2": + return ({"path": "deploy/aws_ec2/main.tf", "content": _AWS_EC2_TF}, + "terraform apply -auto-approve") + if target == "aws_s3_static": + return ({"path": "deploy/aws_s3_static/main.tf", "content": _AWS_S3_STATIC_TF}, + "aws s3 sync ./dist s3://$BUCKET --delete") + if target == "azure_containerapps": + return ({"path": "deploy/azure_containerapps/main.bicep", "content": _AZURE_CONTAINERAPPS}, + "az deployment group create -g rg-prod -f main.bicep") + if target == "azure_appservice": + return ({"path": "deploy/azure_appservice/main.bicep", "content": _AZURE_APPSERVICE}, + "az webapp up --runtime PYTHON:3.12 --sku B1") + if target == "gcp_gce": + return ({"path": "deploy/gcp_gce/main.tf", "content": _GCP_GCE_TF}, + "terraform apply -auto-approve") + return ({"path": "deploy/gcp_cloudrun/main.tf", "content": _GCP_CLOUDRUN_TF}, + "gcloud run deploy nexus-api --source . --region asia-southeast1") + + +_AWS_LAMBDA_SAM = '''# AWS SAM template — Lambda + API Gateway +AWSTemplateFormatVersion: "2010-09-09" +Transform: AWS::Serverless-2016-10-31 +Globals: + Function: + Runtime: python3.12 + MemorySize: 512 + Timeout: 30 + Tracing: Active +Resources: + NexusApi: + Type: AWS::Serverless::Function + Properties: + CodeUri: ../src + Handler: app.handler + Policies: + - AWSLambdaBasicExecutionRole + - DynamoDBCrud: { TableName: !Ref NexusTable } + Environment: + Variables: { LOG_LEVEL: INFO } + Events: + Api: + Type: Api + Properties: + Path: /{proxy+} + Method: ANY + NexusTable: + Type: AWS::DynamoDB::Table + Properties: + BillingMode: PAY_PER_REQUEST + AttributeDefinitions: + - { AttributeName: pk, AttributeType: S } + KeySchema: + - { AttributeName: pk, KeyType: HASH } +Outputs: + ApiUrl: { Value: !Sub "https://${ServerlessRestApi}.execute-api.${AWS::Region}.amazonaws.com/Prod" } +''' + +_AWS_ECS_TF = '''# AWS ECS Fargate + ALB +terraform { + required_version = ">= 1.7" + required_providers { aws = { source = "hashicorp/aws", version = "~> 5.0" } } +} +resource "aws_ecs_cluster" "nexus" { name = "nexus-cluster" } +resource "aws_ecs_task_definition" "nexus" { + family = "nexus-api" + requires_compatibilities = ["FARGATE"] + network_mode = "awsvpc" + cpu = "512" + memory = "1024" + container_definitions = jsonencode([{ + name = "api" + image = "ghcr.io/nexus/api:0.3.0" + portMappings = [{ containerPort = 8000 }] + logConfiguration = { logDriver = "awslogs", + options = { "awslogs-group" = "/ecs/nexus", "awslogs-region" = "ap-southeast-1" } } + }]) +} +resource "aws_ecs_service" "nexus" { + name = "nexus-api" + cluster = aws_ecs_cluster.nexus.id + task_definition = aws_ecs_task_definition.nexus.arn + desired_count = 2 + launch_type = "FARGATE" + network_configuration { + subnets = module.vpc.private_subnets + security_groups = [aws_security_group.nexus.id] + } +} +''' + +_AWS_EC2_TF = '''# AWS EC2 with EIP + user_data +terraform { + required_version = ">= 1.7" + required_providers { aws = { source = "hashicorp/aws", version = "~> 5.0" } } +} +resource "aws_instance" "nexus" { + ami = "ami-0abc1234def56789" + instance_type = "t3.small" + vpc_security_group_ids = [aws_security_group.nexus.id] + iam_instance_profile = aws_iam_instance_profile.nexus.name + user_data = file("deploy/aws_ec2/userdata.sh") + tags = { Name = "nexus-api", Env = "prod" } +} +resource "aws_eip" "nexus" { + instance = aws_instance.nexus.id + domain = "vpc" +} +''' + +_AWS_S3_STATIC_TF = '''# S3 + CloudFront static website +terraform { + required_version = ">= 1.7" + required_providers { aws = { source = "hashicorp/aws", version = "~> 5.0" } } +} +resource "aws_s3_bucket" "static" { bucket = "nexus-static-prod" } +resource "aws_s3_bucket_website_configuration" "static" { + bucket = aws_s3_bucket.static.id + index_document { suffix = "index.html" } + error_document { key = "404.html" } +} +resource "aws_cloudfront_distribution" "cdn" { + origin { + domain_name = aws_s3_bucket_website_configuration.static.website_endpoint + origin_id = "s3-nexus" + custom_origin_config { origin_protocol_policy = "http-only" } + } + enabled = true + is_ipv6_enabled = true + default_cache_behavior { + target_origin_id = "s3-nexus" + viewer_protocol_policy = "redirect-to-https" + allowed_methods = ["GET", "HEAD"] + cached_methods = ["GET", "HEAD"] + forwarded_values { query_string = false; cookies { forward = "none" } } + min_ttl = 0; default_ttl = 3600; max_ttl = 86400 + } + restrictions { geo_restriction { restriction_type = "none" } } + viewer_certificate { cloudfront_default_certificate = true } +} +''' + +_GCP_CLOUDRUN_TF = '''# GCP Cloud Run service +terraform { + required_version = ">= 1.7" + required_providers { google = { source = "hashicorp/google", version = "~> 5.0" } } +} +resource "google_cloud_run_service" "nexus" { + name = "nexus-api" + location = "asia-southeast1" + template { + spec { + container_concurrency = 80 + timeout_seconds = 300 + containers { + image = "gcr.io/PROJECT/nexus-api:0.3.0" + env { name = "LOG_LEVEL"; value = "INFO" } + resources { + limits = { cpu = "1000m", memory = "1Gi" } + } + } + } + } + traffic { percent = 100; latest_revision = true } + autogenerate_revision_name = true +} +resource "google_cloud_run_service_iam_member" "public" { + service = google_cloud_run_service.nexus.name + location = google_cloud_run_service.nexus.location + role = "roles/run.invoker" + member = "allUsers" +} +''' + +_GCP_GCE_TF = '''# GCP Compute Engine with startup script +terraform { + required_version = ">= 1.7" + required_providers { google = { source = "hashicorp/google", version = "~> 5.0" } } +} +resource "google_compute_instance" "nexus" { + name = "nexus-api" + machine_type = "e2-small" + zone = "asia-southeast1-a" + boot_disk { + initialize_params { image = "debian-cloud/debian-12" } + } + network_interface { + network = "default" + access_config {} # ephemeral IP + } + metadata = { startup-script = file("deploy/gcp_gce/startup.sh") } + tags = ["nexus", "http-server"] + service_account { + scopes = ["cloud-platform"] + } +} +''' + +_AZURE_CONTAINERAPPS = '''// Azure Container Apps (Bicep) +param location string = 'southeastasia' +param imageName string = 'ghcr.io/nexus/api:0.3.0' + +resource managedEnv 'Microsoft.App/managedEnvironments@2024-03-01' = { + name: 'nexus-env' + location: location +} + +resource containerApp 'Microsoft.App/containerApps@2024-03-01' = { + name: 'nexus-api' + location: location + properties: { + managedEnvironmentId: managedEnv.id + configuration: { + activeRevisionsMode: 'Single' + ingress: { + external: true + targetPort: 8000 + traffic: [{ weight: 100, latestRevision: true }] + allowInsecure: false + } + secrets: [ + { name: 'api-key', value: '@Microsoft.KeyVault(VaultName=nexus-kv;SecretName=ApiKey)' } + ] + } + template: { + containers: [ + { + name: 'api' + image: imageName + env: [ + { name: 'LOG_LEVEL', value: 'INFO' } + ] + resources: { cpu: json('1.0'), memory: '1.0Gi' } + } + ] + scale: { minReplicas: 1, maxReplicas: 10 } + } + } +} +''' + +_AZURE_APPSERVICE = '''// Azure App Service (Linux, Python) — Bicep +param location string = 'southeastasia' +param sku string = 'B1' + +resource plan 'Microsoft.Web/serverfarms@2023-12-01' = { + name: 'nexus-plan' + location: location + sku: { name: sku, tier: 'Basic' } + properties: { reserved: true } // Linux +} + +resource app 'Microsoft.Web/sites@2023-12-01' = { + name: 'nexus-api' + location: location + properties: { + serverFarmId: plan.id + siteConfig: { + linuxFxVersion: 'PYTHON|3.12' + appCommandLine: 'gunicorn -w 4 -b 0.0.0.0:8000 app:main' + alwaysOn: true + } + } + identity: { type: 'SystemAssigned' } +} +''' diff --git a/nexus/skills/clustering_analysis.py b/nexus/skills/clustering_analysis.py new file mode 100644 index 0000000000000000000000000000000000000000..15c773c49c74c79777f5f7a0c93446efc823ae9b --- /dev/null +++ b/nexus/skills/clustering_analysis.py @@ -0,0 +1,238 @@ +"""Clustering Analysis Skill - KMeans / DBSCAN / Hierarchical pipeline. + +Sinh pipeline clustering hoàn chỉnh: feature scaling, elbow + silhouette +chọn K, fit KMeans / DBSCAN / Agglomerative, đánh giá (silhouette, Davies-Bouldin, +Calinski-Harabasz), và visualization (PCA 2D scatter + dendrogram). + +Author: Hieu Louis (2026) +""" +from __future__ import annotations + +from typing import Dict, List + +from .base import Skill, SkillCategory, SkillContext, SkillPriority, SkillResult + + +CLUSTERING_CODE = '''"""Clustering pipeline / Pipeline phân cụm.""" +from __future__ import annotations +from typing import Dict, List, Tuple +import numpy as np +import pandas as pd +from sklearn.preprocessing import StandardScaler +from sklearn.cluster import KMeans, DBSCAN, AgglomerativeClustering +from sklearn.mixture import GaussianMixture +from sklearn.metrics import ( + silhouette_score, silhouette_samples, + davies_bouldin_score, calinski_harabasz_score, +) +from sklearn.decomposition import PCA +from scipy.cluster.hierarchy import linkage, dendrogram + + +def scale(X: np.ndarray) -> np.ndarray: + """Robust scaling cho clustering / Chuẩn hóa trước khi cluster.""" + return StandardScaler().fit_transform(X) + + +def find_best_k(X: np.ndarray, k_range: range = range(2, 11)) -> Tuple[int, Dict[int, float]]: + """Elbow + silhouette để chọn K / Chọn K bằng silhouette.""" + scores: Dict[int, float] = {} + for k in k_range: + km = KMeans(n_clusters=k, n_init=10, random_state=42).fit(X) + scores[k] = float(silhouette_score(X, km.labels_)) + best_k = max(scores, key=scores.get) + return best_k, scores + + +def kmeans_cluster(X: np.ndarray, k: int, random_state: int = 42) -> Dict[str, object]: + model = KMeans(n_clusters=k, n_init=10, random_state=random_state) + labels = model.fit_predict(X) + return { + "model": model, "labels": labels, + "centroids": model.cluster_centers_, + "inertia": float(model.inertia_), + "silhouette": float(silhouette_score(X, labels)), + } + + +def dbscan_cluster(X: np.ndarray, eps: float = 0.5, min_samples: int = 5) -> Dict[str, object]: + """DBSCAN — auto-detect số cụm, đánh dấu noise (-1).""" + model = DBSCAN(eps=eps, min_samples=min_samples, n_jobs=-1) + labels = model.fit_predict(X) + n_clusters = len(set(labels)) - (1 if -1 in labels else 0) + n_noise = int((labels == -1).sum()) + # Silhouette chỉ tính khi có ≥ 2 cụm thực sự + sil = float(silhouette_score(X, labels)) if n_clusters >= 2 else None + return { + "model": model, "labels": labels, + "n_clusters": n_clusters, "n_noise": n_noise, + "silhouette": sil, + } + + +def agglomerative_cluster(X: np.ndarray, k: int, linkage: str = "ward") -> Dict[str, object]: + model = AgglomerativeClustering(n_clusters=k, linkage=linkage) + labels = model.fit_predict(X) + return { + "model": model, "labels": labels, + "silhouette": float(silhouette_score(X, labels)), + } + + +def gaussian_mixture_cluster(X: np.ndarray, k: int, random_state: int = 42) -> Dict[str, object]: + """GMM — soft clustering, trả về probabilities / Phân cụm mềm.""" + model = GaussianMixture(n_components=k, covariance_type="full", + random_state=random_state, n_init=10) + labels = model.fit_predict(X) + return { + "model": model, "labels": labels, + "proba": model.predict_proba(X), + "bic": float(model.bic(X)), + "aic": float(model.aic(X)), + "silhouette": float(silhouette_score(X, labels)), + } + + +def evaluate(X: np.ndarray, labels: np.ndarray) -> Dict[str, float]: + """Đánh giá clustering khi không có ground-truth.""" + if len(set(labels)) < 2: + return {"silhouette": -1.0, "davies_bouldin": float("inf"), "calinski_harabasz": 0.0} + return { + "silhouette": float(silhouette_score(X, labels)), + "davies_bouldin": float(davies_bouldin_score(X, labels)), # lower = better + "calinski_harabasz": float(calinski_harabasz_score(X, labels)), # higher = better + } + + +def visualize_pca(X: np.ndarray, labels: np.ndarray, title: str = "Clusters (PCA 2D)"): + """PCA 2D scatter tô màu theo cluster / Vẽ PCA.""" + import matplotlib.pyplot as plt + X2 = PCA(n_components=2).fit_transform(scale(X)) + plt.figure(figsize=(8, 5)) + plt.scatter(X2[:, 0], X2[:, 1], c=labels, cmap="tab10", s=12, alpha=0.8) + plt.colorbar(label="cluster") + plt.title(title) + plt.xlabel("PC1"); plt.ylabel("PC2") + plt.tight_layout() + return plt.gcf() + + +def plot_dendrogram(X: np.ndarray, method: str = "ward"): + import matplotlib.pyplot as plt + Z = linkage(scale(X), method=method) + plt.figure(figsize=(10, 5)) + dendrogram(Z, truncate_mode="level", p=5) + plt.title(f"Hierarchical Dendrogram ({method})") + plt.tight_layout() + return plt.gcf() +''' + +STRATEGY_GUIDE = """ +Clustering Strategy Guide / Hướng dẫn chiến lược clustering +============================================================ +1. Preprocess: + - Handle missing values & encode categoricals (OneHot / TargetEncoder). + - Scale features (StandardScaler / RobustScaler) — clustering is distance-based. + - For high-D: reduce first (PCA / UMAP) to combat curse of dimensionality. + +2. Choose algorithm: + - KMeans : spherical clusters, large N, K known + - GMM : elliptical clusters, need soft assignment + - DBSCAN : arbitrary shapes, density-aware, auto-K, robust to noise + - Agglomerative: small N, want dendrogram / hierarchy + - HDBSCAN : variable-density clusters (better than DBSCAN) + - Spectral : graph-based, non-convex clusters + +3. Select K (when needed): + - Elbow on inertia + - Silhouette score (maximize) + - Gap statistic + - Davies-Bouldin (minimize) / Calinski-Harabasz (maximize) + - Domain interpretation + +4. Evaluate (unsupervised): + - Silhouette ∈ [-1, 1] — higher = better separated + - Davies-Bouldin — lower = better + - Calinski-Harabasz — higher = better + +5. Interpret: + - Profile each cluster (mean per feature) + - Visualize via PCA / t-SNE / UMAP 2D projection +""" + + +class ClusteringAnalysisSkill(Skill): + """Sinh clustering pipeline (KMeans/DBSCAN/Hierarchical/GMM) + viz + eval.""" + + category = SkillCategory.ML + priority = SkillPriority.MEDIUM + keywords: List[str] = [ + "cluster", "clustering", "kmeans", "k-means", "dbscan", "hdbscan", + "hierarchical", "agglomerative", "gmm", "gaussian mixture", + "silhouette", "segmentation", "kmeans++", + ] + examples = [ + "Cluster customers với KMeans", + "Tìm số cụm tối ưu bằng silhouette", + "DBSCAN để detect clusters có hình dạng bất kỳ", + ] + + @property + def name(self) -> str: + return "clustering_analysis" + + @property + def description(self) -> str: + return ( + "Sinh pipeline clustering: scaling, KMeans/DBSCAN/Agglomerative/GMM, " + "chọn K (elbow + silhouette), đánh giá (silhouette/DB/CH) + PCA viz." + ) + + def can_handle(self, prompt: str, context: SkillContext = None) -> float: + prompt_lower = prompt.lower() + score = 0.0 + for kw in self.keywords: + if kw in prompt_lower: + score += 0.13 + return min(1.0, score) + + def execute(self, context: SkillContext) -> SkillResult: + prompt_lower = (context.prompt or "").lower() + if "dbscan" in prompt_lower or "hdbscan" in prompt_lower: + recommended = "dbscan" + elif "hierarchical" in prompt_lower or "agglomerative" in prompt_lower or "dendrogram" in prompt_lower: + recommended = "agglomerative" + elif "gmm" in prompt_lower or "gaussian mixture" in prompt_lower: + recommended = "gmm" + else: + recommended = "kmeans" + + artifacts: List[Dict[str, str]] = [ + {"name": "clustering_pipeline.py", "language": "python", "content": CLUSTERING_CODE}, + {"name": "CLUSTERING_STRATEGY.md", "language": "markdown", "content": STRATEGY_GUIDE}, + ] + + return SkillResult( + success=True, + output=( + f"[clustering_analysis] recommended={recommended}\n" + f"Generated pipeline: scaling, K-selection, 4 algorithms, " + f"3 evaluation metrics + PCA/dendrogram viz." + ), + artifacts=artifacts, + suggestions=[ + "Always scale features before clustering (distance-based methods)", + "For high-D data, try PCA / UMAP first to reduce dimensions", + "Profile each cluster (mean per feature) to give business meaning", + "Compare KMeans vs HDBSCAN — HDBSCAN handles variable density better", + "Visualize with both PCA (preserve variance) AND t-SNE/UMAP (preserve locality)", + ], + metadata={ + "skill": self.name, + "recommended_algorithm": recommended, + "algorithms_available": ["kmeans", "dbscan", "agglomerative", "gmm"], + "evaluation_metrics": ["silhouette", "davies_bouldin", "calinski_harabasz"], + "version": self.version, + "author": self.author, + }, + ) diff --git a/nexus/skills/code_completion.py b/nexus/skills/code_completion.py new file mode 100644 index 0000000000000000000000000000000000000000..82f45ec69766e764b923bc29165db2651b3c338c --- /dev/null +++ b/nexus/skills/code_completion.py @@ -0,0 +1,164 @@ +"""Code Completion Skill - Hoàn thành code kiểu Copilot. + +Cung cấp chiến lược completion: context-aware, type-aware, +multi-line completion, Fill-In-the-Middle (FIM), và example artifact. + +Author: Hieu Louis (2026) +""" +from __future__ import annotations + +from typing import Dict, List, Optional + +from .base import Skill, SkillContext, SkillCategory, SkillPriority, SkillResult + + +class CodeCompletionSkill(Skill): + """Hoàn thành code dựa trên context (prefix + suffix + imports).""" + + category = SkillCategory.CODE + priority = SkillPriority.HIGH + keywords: List[str] = [ + "complete", "autocomplete", "copilot", "snippet", + "hoàn thành", "tự động hoàn thành", "fill in", + "infill", "continue code", "next line", + "intellisense", "suggest code", "complete this", + ] + examples = [ + "Complete this function: def factorial(n):", + "Autocomplete the boilerplate for a FastAPI route", + "Copilot-style complete this React component", + ] + + @property + def name(self) -> str: + return "code_completion" + + @property + def description(self) -> str: + return ( + "Hoàn thành code kiểu Copilot: line, block, function-level. " + "Hỗ trợ FIM (Fill-In-the-Middle), context-aware, type-aware." + ) + + def can_handle(self, prompt: str, context: SkillContext = None) -> float: + prompt_lower = prompt.lower() + score = 0.0 + for kw in self.keywords: + if kw in prompt_lower: + score += 0.18 + # Phát hiện dangling code / detect dangling code markers + dangling_markers = ["def ", "function ", "class ", "func ", "fn ", "=>", "{"] + if any(m in prompt for m in dangling_markers) and not prompt.rstrip().endswith((";", "}")): + score += 0.2 + # Cursor markers + if "<|cursor|>" in prompt or "" in prompt or "[[cursor]]" in prompt: + score += 0.4 + return min(1.0, score) + + def execute(self, context: SkillContext) -> SkillResult: + lang = context.language or "python" + return SkillResult( + success=True, + output=( + f"[CodeCompletion/{lang}] FIM-style completion ready. " + f"Producing prefix-suffix-aware suggestion." + ), + artifacts=[ + {"path": "completion/example_completion.txt", "content": _EXAMPLE_COMPLETION}, + {"path": "completion/strategy.md", "content": _COMPLETION_STRATEGY}, + ], + metadata={ + "skill": self.name, + "language": lang, + "modes": { + "line": "single line, no newline insertion", + "block": "multi-line, balanced brackets", + "function": "complete function body from signature", + "file": "scaffold entire file from description", + }, + "fim_format": { + "prompt_template": "{prefix}{suffix}", + "note": "FIM tokens let the model leverage suffix context for mid-line completion", + }, + "context_window_strategy": { + "imports": "always include (1k tokens)", + "type_defs": "include if referenced in prefix", + "same_file_functions": "top-K by retrieval over embeddings", + "recent_edits": "include if within 50 lines of cursor", + "git_diff": "include hunk headers for stylistic consistency", + }, + "ranking_features": [ + "BM25 against project symbols", + "embedding cosine similarity", + "tree-sitter scope awareness", + "type compatibility (mypy/pyright)", + "indentation match", + ], + "safety": { + "secrets_filter": "block completion containing API keys / passwords", + "license_check": "flag verbatim copies of GPL code (>20 token match)", + "syntax_check": "reject if tree-sitter parse fails", + }, + }, + suggestions=[ + "Place cursor marker <|cursor|> exactly where completion should start", + "Provide 3-5 lines of prefix context for best results", + "Specify language and language version explicitly", + "For multi-line completion, indicate desired length (e.g. ~10 lines)", + ], + ) + + +_EXAMPLE_COMPLETION = '''# Example FIM-style completion (language: python) + +# --- Prefix --- +# def quicksort(arr: list[int]) -> list[int]: +# """Sort arr via quicksort, return new list.""" +# if len(arr) <= 1: +# return arr +# pivot = arr[len(arr) // 2] +# <|cursor|> +# --- Suffix --- +# return arr + +# --- Suggested completion --- + left = [x for x in arr if x < pivot] + middle = [x for x in arr if x == pivot] + right = [x for x in arr if x > pivot] + return quicksort(left) + middle + quicksort(right) + +# Confidence: 0.92 | Type-checked: OK | Style: matches PEP-8 +''' + + +_COMPLETION_STRATEGY = """# Code Completion Strategy + +## 1. Context Assembly +- Collect: imports, type definitions, surrounding scope, recent edits. +- Rank candidate context by BM25 + embedding similarity + scope (tree-sitter). + +## 2. FIM (Fill-In-the-Middle) +- Use prefix + suffix tokens to complete mid-line code. +- Critical for partial-line edits, parameter lists, and conditional branches. + +## 3. Candidate Generation +- Generate K=4 candidates (temperature=0.2 for code). +- Nucleus sampling (top_p=0.95) + repetition penalty 1.1. + +## 4. Ranking & Filtering +- syntax_valid (tree-sitter parse) — must pass +- type_check (pyright/mypy for Python) — boost score +- indentation_match (cursor column) — boost score +- secrets_filter — drop candidate +- license_check — flag if verbatim match > 20 tokens + +## 5. Post-processing +- Trim trailing whitespace. +- Balance unbalanced brackets if mode=block. +- Re-indent to match cursor. +- Strip duplicate leading lines already present in prefix. + +## 6. Telemetry (opt-in) +- Log acceptance/rejection, edit distance, latency. +- DO NOT log source code itself, only anonymized metrics. +""" diff --git a/nexus/skills/code_complexity_analysis.py b/nexus/skills/code_complexity_analysis.py new file mode 100644 index 0000000000000000000000000000000000000000..9fda31ab15d0fb48ad2e25cca5661f279eeb0227 --- /dev/null +++ b/nexus/skills/code_complexity_analysis.py @@ -0,0 +1,294 @@ +"""Code Complexity Analysis Skill - Phân tích độ phức tạp. + +Tính Cyclomatic (McCabe) và Cognitive Complexity (SonarSource), +với example calculation cho từng loại. + +Author: Hieu Louis (2026) +""" +from __future__ import annotations + +from typing import Dict, List + +from .base import Skill, SkillContext, SkillCategory, SkillPriority, SkillResult + + +class CodeComplexitySkill(Skill): + """Tính cyclomatic + cognitive complexity, suggest refactors.""" + + category = SkillCategory.CODE + priority = SkillPriority.MEDIUM + keywords: List[str] = [ + "cyclomatic complexity", "cognitive complexity", "complexity", + "mccabe", "code complexity", "độ phức tạp", + "function complexity", "branch complexity", + "too complex", "complex function", + ] + examples = [ + "Calculate cyclomatic complexity of this function", + "Why is this function rated 'complex' by SonarQube?", + "Reduce cognitive complexity of this method", + ] + + @property + def name(self) -> str: + return "code_complexity" + + @property + def description(self) -> str: + return ( + "Tính cyclomatic (McCabe) + cognitive (SonarSource) complexity. " + "Suggest refactors: extract method, guard clauses, polymorphism." + ) + + def can_handle(self, prompt: str, context: SkillContext = None) -> float: + prompt_lower = prompt.lower() + score = 0.0 + for kw in self.keywords: + if kw in prompt_lower: + score += 0.18 + if "def " in prompt or "function " in prompt: + score += 0.1 + return min(1.0, score) + + def execute(self, context: SkillContext) -> SkillResult: + return SkillResult( + success=True, + output="[CodeComplexity] McCabe + cognitive complexity calculator ready.", + artifacts=[ + {"path": "complexity/calculator.py", "content": _COMPLEXITY_CALCULATOR}, + {"path": "complexity/example.md", "content": _EXAMPLE_CALCULATION}, + ], + metadata={ + "skill": self.name, + "metrics": { + "cyclomatic": { + "definition": "M = E - N + 2P (Edges - Nodes + 2*Connected Components)", + "shortcut": "M = decision_points + 1", + "decision_points": ["if", "elif", "for", "while", "except", "and", "or", + "ternary", "case/default"], + "thresholds": { + "low": "<= 5", + "moderate": "6 - 10", + "high": "11 - 20", + "very_high": "21 - 50", + "untestable": "> 50", + }, + }, + "cognitive": { + "definition": "SonarSource metric — penalizes nesting + recursion + breaks", + "increments": [ + "+1 per if/else/for/while/except/case", + "+1 per nesting level (compound cost)", + "+1 per boolean op (and/or/not)", + "+1 per jump (break/continue/return inside loop)", + "+1 per recursion (caller == callee)", + "+1 per goto-like pattern", + ], + "thresholds": { + "low": "<= 5", + "moderate": "6 - 10", + "high": "11 - 20", + "very_high": "21 - 30", + "untestable": "> 30", + }, + }, + "halstead": "Difficulty / Effort / Volume (rarely used in practice)", + "npath": "Number of independent paths — exponential in branches", + }, + "refactor_patterns": [ + "Extract Method (split large function)", + "Replace Conditional with Polymorphism (if-elif ladder -> strategy)", + "Decompose Conditional (long boolean expr -> named predicate)", + "Guard Clauses (early return replaces nested if-else)", + "Replace Nested Conditionals with State/Strategy", + "Compose Method (sequence of intention-revealing calls)", + ], + "tooling": { + "python": "radon cc (cyclomatic), radon mi (maintainability), xenon (CI)", + "javascript": "escomplex, typhonjs-escomplex", + "java": "PMD, SonarQube", + "go": "gocyclo (cyclomatic only)", + "rust": "rust-code-analysis (both metrics)", + }, + "ci_thresholds": { + "block_pr": "cyclomatic > 15 OR cognitive > 20", + "warn": "cyclomatic > 10 OR cognitive > 15", + "trend": "Track average per file; fail regression > 10%", + }, + }, + suggestions=[ + "Specify which metric (cyclomatic / cognitive / both)", + "Provide code in fenced block for accurate analysis", + "Ask for refactor suggestions if complexity > threshold", + ], + ) + + +_COMPLEXITY_CALCULATOR = '''"""Cyclomatic + Cognitive Complexity calculator. + +Author: Hieu Louis (2026) +""" +from __future__ import annotations +import ast +from dataclasses import dataclass + + +@dataclass +class ComplexityResult: + cyclomatic: int + cognitive: int + decision_points: int + nesting_max: int + rating: str # "low" | "moderate" | "high" | "very_high" | "untestable" + + +def analyze(func: ast.FunctionDef) -> ComplexityResult: + visitor = _ComplexityVisitor(func.name) + visitor.visit(func) + cyclo = visitor.decision_points + 1 + cognitive = visitor.cognitive + nesting_max = visitor.max_nesting + rating = _rate(cyclo, cognitive) + return ComplexityResult( + cyclomatic=cyclo, + cognitive=cognitive, + decision_points=visitor.decision_points, + nesting_max=nesting_max, + rating=rating, + ) + + +# Cyclomatic: count decision points +# Cognitive: SonarSource algorithm (penalize nesting + recursion + jumps) + + +class _ComplexityVisitor(ast.NodeVisitor): + DECISION_NODES = ( + ast.If, ast.For, ast.AsyncFor, ast.While, + ast.ExceptHandler, ast.BoolOp, ast.IfExp, + ) + + def __init__(self, func_name: str) -> None: + self.func_name = func_name + self.decision_points = 0 + self.cognitive = 0 + self.nesting = 0 + self.max_nesting = 0 + self.in_loop = False + + def _visit_decision(self, node): + self.decision_points += 1 + self.cognitive += self.nesting + 1 + self.nesting += 1 + self.max_nesting = max(self.max_nesting, self.nesting) + self.generic_visit(node) + self.nesting -= 1 + + def visit_BoolOp(self, node: ast.BoolOp) -> None: + # Each additional operand in `and`/`or` is +1 + self.decision_points += max(0, len(node.values) - 1) + self.cognitive += max(0, len(node.values) - 1) + self.generic_visit(node) + + visit_If = _visit_decision + visit_For = _visit_decision + visit_AsyncFor = _visit_decision + visit_While = _visit_decision + visit_ExceptHandler = _visit_decision + + def visit_IfExp(self, node: ast.IfExp) -> None: + self.decision_points += 1 + self.cognitive += 1 + self.generic_visit(node) + + def visit_Break(self, node: ast.Break) -> None: + if self.in_loop: + self.cognitive += 1 + self.generic_visit(node) + + def visit_Continue(self, node: ast.Continue) -> None: + if self.in_loop: + self.cognitive += 1 + self.generic_visit(node) + + def visit_FunctionDef(self, node: ast.FunctionDef) -> None: + if node.name == self.func_name: + self.cognitive += 1 # recursion penalty + else: + self._visit_decision(node) + + visit_AsyncFunctionDef = visit_FunctionDef + + +def _rate(cyclo: int, cognitive: int) -> str: + if cyclo <= 5 and cognitive <= 5: + return "low" + if cyclo <= 10 and cognitive <= 10: + return "moderate" + if cyclo <= 20 and cognitive <= 20: + return "high" + if cyclo <= 50 and cognitive <= 30: + return "very_high" + return "untestable" +''' + + +_EXAMPLE_CALCULATION = '''# Example: Cyclomatic + Cognitive Complexity Calculation + +## Sample Code +```python +def process(items, flag): + result = [] + for item in items: # cyclomatic +1, cognitive +1 + if item.is_valid and flag: # cyclomatic +1 (if) +1 (and), cognitive +2 (nested) +1 (and) + if item.priority > 5: # cyclomatic +1, cognitive +3 (doubly nested) + result.append(item) + else: + continue # cognitive +1 (jump in loop) + elif item.is_optional: # cyclomatic +1 (elif), cognitive +2 + result.append(item) + return result +``` + +## Cyclomatic Complexity (McCabe) +Decision points counted: +- `for` ... 1 +- `if` ... 1 +- `and` ... 1 +- `if` (nested) ... 1 +- `elif` ... 1 +Total decision_points = 5 + +`M = decision_points + 1 = 6` + +Rating: **moderate** + +## Cognitive Complexity (SonarSource) +- `for` at nesting 0: +1 (nesting 0 + base 1) +- `if` at nesting 1: +2 (nesting 1 + base 1) +- `and` operand: +1 +- nested `if` at nesting 2: +3 (nesting 2 + base 1) +- `continue` (jump in loop): +1 +- `elif` at nesting 1: +2 (nesting 1 + base 1) +Total cognitive = 1 + 2 + 1 + 3 + 1 + 2 = **10** + +Rating: **moderate** (close to high boundary 11) + +## Refactor Suggestions +1. **Extract Method**: pull nested `if item.priority > 5` into `_should_include(item)`. +2. **Guard Clause**: replace `elif` with early `continue` to flatten structure. +3. **Replace Conditional with Strategy** if `flag`/`priority` combos grow. + +## Refactored (target: cyclo <= 4, cognitive <= 5) +```python +def process(items, flag): + return [it for it in items if _should_keep(it, flag)] + +def _should_keep(item, flag): + if not (item.is_valid and flag): + return item.is_optional + return item.priority > 5 +``` +- `process`: cyclo=1, cognitive=1 +- `_should_keep`: cyclo=2, cognitive=3 +''' diff --git a/nexus/skills/code_dead_code_analysis.py b/nexus/skills/code_dead_code_analysis.py new file mode 100644 index 0000000000000000000000000000000000000000..0a7e45637a93949902f3c478e94a1d9522043e36 --- /dev/null +++ b/nexus/skills/code_dead_code_analysis.py @@ -0,0 +1,310 @@ +"""Dead Code Analysis Skill - Phát hiện dead / unreachable code. + +Sử dụng control-flow analysis (CFG), use-def chains, static reachability, +và call-graph traversal để phát hiện: +- Unreachable statements +- Unused functions / variables / imports +- Unused private methods +- Unreachable branches (always-true/false conditions) + +Author: Hieu Louis (2026) +""" +from __future__ import annotations + +from typing import Dict, List + +from .base import Skill, SkillContext, SkillCategory, SkillPriority, SkillResult + + +class DeadCodeAnalysisSkill(Skill): + """Phát hiện dead code: unreachable, unused, never-called.""" + + category = SkillCategory.CODE + priority = SkillPriority.LOW + keywords: List[str] = [ + "dead code", "unused", "unreachable", "never called", + "dead function", "unused import", "unused variable", + "code không dùng", "code chết", "orphan code", + "zombie code", "dead branch", + ] + examples = [ + "Find dead code in this module", + "Detect unused private methods", + "Report unreachable branches after refactor", + ] + + @property + def name(self) -> str: + return "dead_code_analysis" + + @property + def description(self) -> str: + return ( + "Phát hiện dead code qua CFG + use-def chains + call-graph: " + "unreachable statements, unused symbols, never-called functions." + ) + + def can_handle(self, prompt: str, context: SkillContext = None) -> float: + prompt_lower = prompt.lower() + score = 0.0 + for kw in self.keywords: + if kw in prompt_lower: + score += 0.2 + return min(1.0, score) + + def execute(self, context: SkillContext) -> SkillResult: + return SkillResult( + success=True, + output="[DeadCodeAnalysis] CFG + use-def + call-graph analysis ready.", + artifacts=[ + {"path": "dead_code/analyzer.py", "content": _DEAD_CODE_ANALYZER}, + {"path": "dead_code/checklist.md", "content": _DEAD_CODE_CHECKLIST}, + ], + metadata={ + "skill": self.name, + "categories": { + "unreachable_stmt": "Statement after return/raise/break/continue", + "unreachable_branch": "Branch with always-true/false condition", + "unused_local": "Local variable assigned but never read", + "unused_private_method": "Private method never called within module", + "unused_import": "Imported symbol not referenced", + "unreferenced_module": "Module never imported by entry points", + "orphan_file": "File not in build graph / not imported anywhere", + }, + "analysis_phases": [ + "1. Build module-level AST + import graph", + "2. Build call graph (caller -> callee edges)", + "3. Reachability from public entry points (main, exports, tests)", + "4. Per-function CFG: detect unreachable blocks via predecessor analysis", + "5. Use-def chains: variables defined but never used", + "6. Constant propagation: detect always-true/false conditions", + "7. Cross-module: unreferenced modules / orphan files", + ], + "tooling": { + "python": "vulture, pyflakes (F401 unused import), depy (call-graph)", + "javascript": "ts-prune, knip (finds unused exports + files)", + "typescript": "ts-prune, knip", + "go": "deadcode (built into `go tool`)", + "rust": "cargo udeps (needs nightly), cargo machete", + "java": "PMD, IntelliJ 'unused declaration' inspection", + "c++": "cppcheck --enable=unusedFunction", + }, + "false_positive_mitigations": [ + "Reflection / dynamic dispatch (mark @api entries)", + "Metaprogramming (decorators, __all__, exports)", + "String-based dispatch (event handlers, route registration)", + "External entry points (CLI commands, plugin systems)", + "Test-only utilities (keep if covered by tests)", + ], + "ci_integration": { + "fail_on_new": "True — block PRs introducing new dead code", + "allowlist": "Pre-existing dead code tracked in `deadcode-allowlist.yaml`", + "trend_metric": "Track dead_code_lines / total_lines over time", + }, + }, + suggestions=[ + "Provide entry points (main module / CLI) for accurate reachability", + "Mark public API surfaces with @api decorator before scan", + "Allow reflection-heavy modules with explicit allowlist", + ], + ) + + +_DEAD_CODE_ANALYZER = '''"""Dead code analyzer: CFG + use-def + call-graph reachability. + +Author: Hieu Louis (2026) +""" +from __future__ import annotations +import ast +from collections import defaultdict +from dataclasses import dataclass, field +from typing import Dict, List, Set, Tuple + + +@dataclass +class DeadCodeFinding: + kind: str # "unreachable" | "unused_local" | "unused_func" ... + file: str + lineno: int + end_lineno: int + symbol: str + reason: str + + +@dataclass +class AnalysisReport: + findings: List[DeadCodeFinding] = field(default_factory=list) + entry_points: Set[str] = field(default_factory=set) + reachable_funcs: Set[str] = field(default_factory=set) + + @property + def dead_function_count(self) -> int: + return sum(1 for f in self.findings if f.kind == "unused_func") + + @property + def unreachable_lines(self) -> int: + return sum( + f.end_lineno - f.lineno + 1 + for f in self.findings + if f.kind == "unreachable" + ) + + +def analyze(files: List[str], entry_points: Set[str]) -> AnalysisReport: + """Run full dead-code analysis pipeline.""" + report = AnalysisReport(entry_points=entry_points) + + # Phase 1: parse all files into module-level defs + defs: Dict[str, Tuple[str, ast.AST]] = {} + for path in files: + src = open(path, encoding="utf-8").read() + try: + tree = ast.parse(src) + except SyntaxError: + continue + for node in tree.body: + name = getattr(node, "name", None) + if name: + defs[name] = (path, node) + + # Phase 2: build call graph (caller -> callees) + callers: Dict[str, Set[str]] = defaultdict(set) + for name, (path, node) in defs.items(): + for child in ast.walk(node): + if isinstance(child, ast.Call): + callee = _get_callee_name(child) + if callee: + callers[callee].add(name) + + # Phase 3: reachability from entry points + reachable: Set[str] = set() + queue = list(entry_points) + while queue: + fn = queue.pop() + if fn in reachable: + continue + reachable.add(fn) + for caller in callers.get(fn, set()): + if caller not in reachable: + queue.append(caller) + report.reachable_funcs = reachable + + # Phase 4: emit unused functions (private + not reachable) + for name, (path, node) in defs.items(): + is_private = name.startswith("_") or name.islower() + if is_private and name not in reachable and name not in entry_points: + report.findings.append(DeadCodeFinding( + kind="unused_func", + file=path, + lineno=node.lineno, + end_lineno=getattr(node, "end_lineno", node.lineno), + symbol=name, + reason="Private function not reachable from entry points", + )) + + # Phase 5: per-function unreachable statements + for path, node in defs.values(): + if isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef)): + for stmt in _find_unreachable(node): + report.findings.append(DeadCodeFinding( + kind="unreachable", + file=path, + lineno=stmt.lineno, + end_lineno=getattr(stmt, "end_lineno", stmt.lineno), + symbol=_snippet(stmt), + reason="Statement after return/raise/break/continue", + )) + + # Phase 6: unused local variables (use-def) + for path, node in defs.values(): + for unused in _find_unused_locals(node): + report.findings.append(DeadCodeFinding( + kind="unused_local", + file=path, + lineno=unused.lineno, + end_lineno=unused.lineno, + symbol=unused.id, + reason="Local assigned but never read", + )) + + return report + + +def _get_callee_name(call: ast.Call) -> str: + if isinstance(call.func, ast.Name): + return call.func.id + if isinstance(call.func, ast.Attribute): + return call.func.attr + return "" + + +def _find_unreachable(func: ast.FunctionDef) -> List[ast.stmt]: + """Return statements appearing after terminator (return/raise/break/continue).""" + unreachable: List[ast.stmt] = [] + terminated = False + for stmt in func.body: + if terminated: + unreachable.append(stmt) + continue + if isinstance(stmt, (ast.Return, ast.Raise, ast.Break, ast.Continue)): + terminated = True + return unreachable + + +def _find_unused_locals(func: ast.FunctionDef) -> List[ast.Name]: + """Use-def: assigned but never read.""" + assigned: Dict[str, ast.Name] = {} + read: Set[str] = set() + for node in ast.walk(func): + if isinstance(node, ast.Name) and isinstance(node.ctx, ast.Store): + assigned.setdefault(node.id, node) + elif isinstance(node, ast.Name) and isinstance(node.ctx, ast.Load): + read.add(node.id) + return [n for name, n in assigned.items() if name not in read] + + +def _snippet(stmt: ast.stmt) -> str: + """Short text representation of a statement for the report.""" + if isinstance(stmt, ast.Return): + return "return" + if isinstance(stmt, ast.Raise): + return "raise" + if isinstance(stmt, ast.Assign): + return "assign" + return type(stmt).__name__ +''' + + +_DEAD_CODE_CHECKLIST = """# Dead Code Detection Checklist + +## Per-Function (CFG-level) +- [ ] Statement after `return` / `raise` / `break` / `continue`? +- [ ] `if False:` / `if True:` constant-folded branches? +- [ ] `while False:` loop body? +- [ ] `assert False` unreachable successors? +- [ ] Exception handler that never matches raised type? + +## Per-Module (Symbol-level) +- [ ] Private functions (`_foo`) reachable from public entry points? +- [ ] Module-level constants used anywhere? +- [ ] Imported symbols all referenced? +- [ ] Class methods called (or registered as `@property` / `@staticmethod`)? + +## Per-Codebase (Graph-level) +- [ ] All modules reachable from entry points (main / `__init__.py` / CLI)? +- [ ] All public API functions either have callers or are exported in `__all__`? +- [ ] Plugin-style registrations (`@route`, `@click.command`) covered? +- [ ] Test utilities isolated from production code? + +## False Positive Sources +- Reflection: `getattr(obj, "method_name")` +- Dynamic dispatch: registry pattern `REGISTRY["key"]()` +- Serialization: `__init__.py` `__all__` exports +- External API: framework hooks (`pytest fixtures`, `click commands`) +- Type-only imports (TS): `import type { Foo }` — keep for type checks + +## Trend Tracking +- Plot `dead_code_lines / total_lines` weekly. +- Set ceiling: e.g. dead ratio < 5%. +- Auto-file issue when ratio increases > 1% in a sprint. +""" diff --git a/nexus/skills/code_dependency_analysis.py b/nexus/skills/code_dependency_analysis.py new file mode 100644 index 0000000000000000000000000000000000000000..d66b00c1ca54ea089ff05398301971f2022a6324 --- /dev/null +++ b/nexus/skills/code_dependency_analysis.py @@ -0,0 +1,274 @@ +"""Code Dependency Analysis Skill - Phân tích dependency graph. + +Extract import graph, build dependency tree, detect circular dependencies, +compute fan-in / fan-out, và suggest module boundaries. + +Author: Hieu Louis (2026) +""" +from __future__ import annotations + +from typing import Dict, List + +from .base import Skill, SkillContext, SkillCategory, SkillPriority, SkillResult + + +class CodeDependencySkill(Skill): + """Trích dependency graph, phát hiện cycle, tính fan-in/out.""" + + category = SkillCategory.CODE + priority = SkillPriority.MEDIUM + keywords: List[str] = [ + "dependency", "dependencies", "import", "dependency tree", + "depgraph", "import graph", "circular import", + "module dependency", "fan in", "fan out", + "sự phụ thuộc", "đồ thị phụ thuộc", + ] + examples = [ + "Show the dependency graph of this package", + "Find circular imports in the codebase", + "Which modules have the highest fan-in?", + ] + + @property + def name(self) -> str: + return "code_dependency" + + @property + def description(self) -> str: + return ( + "Trích dependency graph: imports, call graph, fan-in/fan-out, " + "circular dependency detection, suggest module boundaries." + ) + + def can_handle(self, prompt: str, context: SkillContext = None) -> float: + prompt_lower = prompt.lower() + score = 0.0 + for kw in self.keywords: + if kw in prompt_lower: + score += 0.15 + return min(1.0, score) + + def execute(self, context: SkillContext) -> SkillResult: + return SkillResult( + success=True, + output="[CodeDependency] Import graph extraction + cycle detection ready.", + artifacts=[ + {"path": "dependency/graph_extractor.py", "content": _GRAPH_EXTRACTOR}, + {"path": "dependency/report_template.md", "content": _REPORT_TEMPLATE}, + ], + metadata={ + "skill": self.name, + "graph_types": { + "import_graph": "module -> set of imported modules (static)", + "call_graph": "function -> set of called functions (intra-procedural)", + "type_graph": "class -> set of referenced types", + "runtime_graph": "actual module loads (instrumented, e.g. sys.modules diff)", + }, + "metrics": { + "fan_in": "Number of modules depending on this one", + "fan_out": "Number of modules this one depends on", + "instability": "I = fan_out / (fan_in + fan_out) — 0 = stable, 1 = unstable", + "abstractness": "A = abstract_classes / total_classes (per module)", + "distance_main_seq": "D = |A + I - 1| — 0 is on the main sequence (good)", + }, + "cycle_detection": [ + "Tarjan SCC (Strongly Connected Components) — O(V+E)", + "DFS with color marking (white/gray/black) — simpler, O(V+E)", + "Johnson's algorithm for enumerating ALL elementary cycles", + ], + "visualization": { + "graphviz": "dot -Tsvg deps.dot -o deps.svg", + "mermaid": "graph TD; A-->B; B-->C;", + "d3": "force-directed layout for interactive exploration", + "cytoscape": "for large graphs (10k+ nodes)", + }, + "tooling": { + "python": "pydeps, snakefood, pyreverse (built-in with pylint)", + "javascript": "madge (CLI + lib, supports circular detection)", + "typescript": "madge, dependency-cruiser (rules-based)", + "go": "go mod graph, goda (rich analysis)", + "rust": "cargo tree, cargo-deny (license/advisory)", + "java": "Maven Enforcer (ban-circular-dependencies), JDeps", + }, + "refactor_targets": [ + "God module: fan_in + fan_out both very high", + "Cycle: A->B->C->A — break with Dependency Inversion (interface in shared module)", + "Leaky abstraction: low-level module imported by high-level (SOLID violation)", + "Dead module: zero fan_in (orphan)", + ], + }, + suggestions=[ + "Provide package root or list of files to scan", + "Specify output format: dot / mermaid / json", + "Run cycle detection if refactoring is planned", + ], + ) + + +_GRAPH_EXTRACTOR = '''"""Dependency graph extractor for Python modules. + +Builds import graph, detects cycles (Tarjan SCC), computes fan-in/out. + +Author: Hieu Louis (2026) +""" +from __future__ import annotations +import ast +import os +from collections import defaultdict +from dataclasses import dataclass, field +from typing import Dict, List, Set, Tuple + + +@dataclass +class DependencyGraph: + edges: Dict[str, Set[str]] = field(default_factory=lambda: defaultdict(set)) + modules: Set[str] = field(default_factory=set) + + def add_edge(self, src: str, dst: str) -> None: + if src != dst: + self.edges[src].add(dst) + self.modules.add(src) + self.modules.add(dst) + + def fan_out(self, module: str) -> int: + return len(self.edges.get(module, set())) + + def fan_in(self, module: str) -> int: + return sum(1 for src, dsts in self.edges.items() if module in dsts) + + def instability(self, module: str) -> float: + fin, fout = self.fan_in(module), self.fan_out(module) + total = fin + fout + return fout / total if total else 0.0 + + +def build_import_graph(root: str, package_name: str) -> DependencyGraph: + """Walk directory, parse each .py, extract import edges.""" + graph = DependencyGraph() + for dirpath, _dirs, files in os.walk(root): + for fname in files: + if not fname.endswith(".py"): + continue + path = os.path.join(dirpath, fname) + module = _path_to_module(os.path.relpath(path, root), package_name) + src = open(path, encoding="utf-8").read() + try: + tree = ast.parse(src) + except SyntaxError: + continue + for node in ast.walk(tree): + for dep in _extract_imports(node, package_name): + graph.add_edge(module, dep) + return graph + + +def _extract_imports(node: ast.AST, package_name: str) -> List[str]: + """Return list of imported module dotted names (only local package).""" + deps: List[str] = [] + if isinstance(node, ast.Import): + for alias in node.names: + if alias.name.startswith(package_name): + deps.append(alias.name) + elif isinstance(node, ast.ImportFrom): + if node.module and node.module.startswith(package_name): + deps.append(node.module) + return deps + + +def _path_to_module(rel_path: str, package_name: str) -> str: + parts = rel_path.replace(os.sep, ".").removesuffix(".py") + if parts.endswith(".__init__"): + parts = parts.removesuffix(".__init__") + return f"{package_name}.{parts}" if parts else package_name + + +def find_cycles(graph: DependencyGraph) -> List[List[str]]: + """Tarjan SCC algorithm — returns list of strongly connected components + of size >= 2 (these contain cycles).""" + index_counter = [0] + stack: List[str] = [] + lowlink: Dict[str, int] = {} + index: Dict[str, int] = {} + on_stack: Dict[str, bool] = {} + sccs: List[List[str]] = [] + + def strongconnect(node: str) -> None: + index[node] = index_counter[0] + lowlink[node] = index_counter[0] + index_counter[0] += 1 + stack.append(node) + on_stack[node] = True + for succ in graph.edges.get(node, set()): + if succ not in index: + strongconnect(succ) + lowlink[node] = min(lowlink[node], lowlink[succ]) + elif on_stack.get(succ): + lowlink[node] = min(lowlink[node], index[succ]) + if lowlink[node] == index[node]: + comp: List[str] = [] + while True: + w = stack.pop() + on_stack[w] = False + comp.append(w) + if w == node: + break + if len(comp) >= 2: + sccs.append(comp) + + for m in graph.modules: + if m not in index: + strongconnect(m) + return sccs + + +def hotspots(graph: DependencyGraph, top_k: int = 10) -> List[Tuple[str, int, int, float]]: + """Return top-K modules by instability — refactor candidates.""" + rows = [ + (m, graph.fan_in(m), graph.fan_out(m), graph.instability(m)) + for m in graph.modules + ] + return sorted(rows, key=lambda r: r[3], reverse=True)[:top_k] +''' + + +_REPORT_TEMPLATE = '''# Dependency Analysis Report + +## Summary +- Modules analyzed: +- Total edges: +- Cycles detected: +- Orphan modules (fan_in=0): + +## Graph Visualization +```dot +digraph deps { + rankdir=LR; + node [shape=box]; + "pkg.api" -> "pkg.service"; + "pkg.service" -> "pkg.repo"; + "pkg.repo" -> "pkg.models"; + "pkg.api" -> "pkg.models"; // shortcut — consider removing +} +``` + +## Cycle Report +``` +Cycle #1 (length 3): + pkg.a -> pkg.b -> pkg.c -> pkg.a + +Suggested fix: extract shared interface into pkg.interfaces, +invert dependency: pkg.a depends on pkg.interfaces, pkg.c implements it. +``` + +## Instability Hotspots (top 10) +| Module | Fan-in | Fan-out | Instability | Notes | +|---------------|--------|---------|-------------|------------------------| +| pkg.api | 0 | 8 | 1.00 | entry point — OK | +| pkg.utils | 14 | 2 | 0.13 | god module — review | +| pkg.models | 22 | 1 | 0.04 | stable foundation — OK | + +## Action Items +- [ ] Break cycle in pkg.a / pkg.b / pkg.c via interface extraction +- [ ] Split pkg.utils (high fan-in + high fan-out = god module) +- [ ] Verify pkg.orphan is truly dead (run dead-code skill) +''' diff --git a/nexus/skills/code_documentation_generation.py b/nexus/skills/code_documentation_generation.py new file mode 100644 index 0000000000000000000000000000000000000000..0e94f0dff80d5599cf1d72ceaeb9d24e3a33b212 --- /dev/null +++ b/nexus/skills/code_documentation_generation.py @@ -0,0 +1,295 @@ +"""Code Documentation Skill - Sinh docstring/comment tự động. + +Hỗ trợ Python (Google/NumPy/Sphinx), JS (JSDoc), TS (TSDoc), Go (godoc), +Rust (rustdoc), Java (Javadoc), với template per style. + +Author: Hieu Louis (2026) +""" +from __future__ import annotations + +from typing import Dict, List + +from .base import Skill, SkillContext, SkillCategory, SkillPriority, SkillResult + + +class CodeDocumentationSkill(Skill): + """Sinh docstring và comment cho function/class/module.""" + + category = SkillCategory.DOCUMENTATION + priority = SkillPriority.MEDIUM + keywords: List[str] = [ + "docstring", "document function", "document class", + "jsdoc", "javadoc", "godoc", "rustdoc", "tsdoc", + "generate docs", "documentation", "tài liệu", + "viết docstring", "comment code", "annotate", + ] + examples = [ + "Generate Google-style docstring for this Python function", + "Write JSDoc for this JavaScript function", + "Document all public methods of this class", + ] + + @property + def name(self) -> str: + return "code_documentation" + + @property + def description(self) -> str: + return ( + "Sinh docstring/comment cho Python (Google/NumPy/Sphinx), " + "JS (JSDoc), TS (TSDoc), Go (godoc), Rust (rustdoc), Java (Javadoc)." + ) + + def can_handle(self, prompt: str, context: SkillContext = None) -> float: + prompt_lower = prompt.lower() + score = 0.0 + for kw in self.keywords: + if kw in prompt_lower: + score += 0.16 + if "def " in prompt or "function " in prompt or "func " in prompt: + score += 0.15 + return min(1.0, score) + + def execute(self, context: SkillContext) -> SkillResult: + lang = (context.language or "python").lower() + return SkillResult( + success=True, + output=f"[CodeDocumentation/{lang}] Docstring templates ready.", + artifacts=[ + {"path": "docs/templates.md", "content": _DOCSTRING_TEMPLATES}, + {"path": "docs/jsdoc_template.md", "content": _JSDOC_TEMPLATE}, + {"path": "docs/strategy.md", "content": _DOC_STRATEGY}, + ], + metadata={ + "skill": self.name, + "language": lang, + "styles": { + "python": ["google", "numpy", "sphinx", "rest"], + "javascript": ["jsdoc"], + "typescript": ["tsdoc (typedoc)"], + "go": ["godoc (no annotations)"], + "rust": ["rustdoc markdown"], + "java": ["javadoc"], + "kotlin": ["kdoc"], + }, + "extraction_targets": [ + "purpose (first sentence)", + "parameters (name, type, meaning, default, constraints)", + "return value (type, meaning, conditions)", + "raises/throws (exception types + when)", + "examples (doctest-runnable when possible)", + "side effects", + "deprecated + replacement", + "see also", + ], + "tooling": { + "python": "Sphinx + autodoc + napoleon + intersphinx", + "js": "TypeDoc (TS) / JSDoc (JS)", + "go": "godoc / pkg.go.dev", + "rust": "cargo doc", + "java": "Javadoc + Maven Javadoc plugin", + }, + "validation": [ + "doctest for Python examples", + "mypy/pyright on type annotations", + "lint: every public symbol has docs (CI check)", + ], + }, + suggestions=[ + "Pick style explicitly: 'google' / 'numpy' / 'sphinx' for Python", + "Ask for doctest-runnable examples when applicable", + "Document exceptions explicitly even if not raised directly", + ], + ) + + +_DOCSTRING_TEMPLATES = '''# Python Docstring Templates + +## Google style + +```python +def compute_discount(cart, customer_tier, coupon=None): + """Compute discount for a cart. + + Applies tiered discount rules based on cart total and customer tier. + Discount is capped at 40% for retail customers. + + Args: + cart: List of (sku, unit_price, quantity) tuples. Must be non-empty. + customer_tier: One of "bronze", "silver", "gold". Case-insensitive. + coupon: Optional coupon code. None for no coupon. + + Returns: + Tuple of (discount_amount, final_total). discount_amount in + [0, cart_subtotal]. final_total is non-negative. + + Raises: + ValueError: If cart is empty or customer_tier is unknown. + CouponExpiredError: If coupon code is past its expiry date. + + Examples: + >>> compute_discount([("A1", 100, 2)], "gold") + (20.0, 180.0) + """ + ... +``` + +## NumPy style + +```python +def compute_discount(cart, customer_tier, coupon=None): + """Compute discount for a cart. + + Applies tiered discount rules based on cart total and customer tier. + + Parameters + ---------- + cart : list[tuple[str, float, int]] + List of (sku, unit_price, quantity) tuples. Must be non-empty. + customer_tier : {"bronze", "silver", "gold"} + Customer loyalty tier. Case-insensitive. + coupon : str, optional + Optional coupon code. None for no coupon. + + Returns + ------- + tuple[float, float] + (discount_amount, final_total). discount_amount in [0, subtotal]. + + Raises + ------ + ValueError + If cart is empty or customer_tier is unknown. + CouponExpiredError + If coupon code is past its expiry date. + + Examples + -------- + >>> compute_discount([("A1", 100, 2)], "gold") + (20.0, 180.0) + """ + ... +``` + +## Sphinx (reST) style + +```python +def compute_discount(cart, customer_tier, coupon=None): + """Compute discount for a cart. + + Applies tiered discount rules based on cart total and customer tier. + + :param cart: List of (sku, unit_price, quantity) tuples. Must be non-empty. + :type cart: list[tuple[str, float, int]] + :param customer_tier: One of "bronze", "silver", "gold". Case-insensitive. + :type customer_tier: str + :param coupon: Optional coupon code. None for no coupon. + :type coupon: str | None + :returns: (discount_amount, final_total). + :rtype: tuple[float, float] + :raises ValueError: If cart is empty or customer_tier is unknown. + :raises CouponExpiredError: If coupon code is past expiry. + + Example:: + + >>> compute_discount([("A1", 100, 2)], "gold") + (20.0, 180.0) + """ + ... +``` +''' + + +_JSDOC_TEMPLATE = '''# JSDoc / TSDoc Template + +```javascript +/** + * Compute discount for a cart. + * + * Applies tiered discount rules based on cart total and customer tier. + * Discount is capped at 40% for retail customers. + * + * @param {Array<{sku: string, unitPrice: number, quantity: number}>} cart + * List of cart items. Must be non-empty. + * @param {"bronze" | "silver" | "gold"} customerTier + * Customer loyalty tier. Case-insensitive. + * @param {string | null} [coupon=null] + * Optional coupon code. Pass null for no coupon. + * @returns {{discountAmount: number, finalTotal: number}} + * Discount amount (0 <= d <= subtotal) and final total. + * @throws {TypeError} If cart is empty. + * @throws {CouponExpiredError} If coupon code is past expiry. + * + * @example + * const { discountAmount, finalTotal } = computeDiscount( + * [{ sku: "A1", unitPrice: 100, quantity: 2 }], + * "gold" + * ); + * // => { discountAmount: 20, finalTotal: 180 } + * + * @see {@link applyCoupon} for coupon resolution logic. + * @since 1.2.0 + * @public + */ +function computeDiscount(cart, customerTier, coupon = null) { + // ... +} +``` + +## TSDoc (TypeScript) variant + +```typescript +/** + * Compute discount for a cart. + * + * @param cart - List of cart items. Must be non-empty. + * @param customerTier - Customer loyalty tier. Case-insensitive. + * @param coupon - Optional coupon code. Pass null for no coupon. + * @returns Discount amount and final total. + * @throws {TypeError} If cart is empty. + * + * @example + * ```ts + * const r = computeDiscount([{ sku: "A1", unitPrice: 100, quantity: 2 }], "gold"); + * ``` + */ +function computeDiscount( + cart: CartItem[], + customerTier: "bronze" | "silver" | "gold", + coupon: string | null = null, +): { discountAmount: number; finalTotal: number } { + // ... +} +``` +''' + + +_DOC_STRATEGY = """# Documentation Generation Strategy + +## Phase 1: Static Extraction +- Parse AST, collect: function signatures, parameter types, return types, + raised exceptions, decorators, class hierarchy. +- Infer types when missing (mypy/pyright inference). + +## Phase 2: Purpose Inference +- Heuristics: function name + first assignment + last return + called functions. +- LLM fallback: ask for one-sentence summary, validate against signature. + +## Phase 3: Parameter Description +- Per parameter: infer from usage (read once? written? returned?). +- Look at type hints + constraint annotations. +- Generate description: " is the : ". + +## Phase 4: Examples +- Generate 1 happy-path + 1 error example. +- Make examples doctest-runnable (Python) or runnable snippets (JS). + +## Phase 5: Style Compliance +- Match existing docstring style in module (auto-detect: google/numpy/sphinx). +- Match indentation, line length, terminology. + +## Phase 6: Validation +- doctest: every `>>>` block must pass. +- darglint / pydocstyle / flake8-docstrings lint. +- Verify all params in signature have `Args:` entries. +""" diff --git a/nexus/skills/code_duplication_detection.py b/nexus/skills/code_duplication_detection.py new file mode 100644 index 0000000000000000000000000000000000000000..91308e5f70e9943b2859c8c82791463a77ad030b --- /dev/null +++ b/nexus/skills/code_duplication_detection.py @@ -0,0 +1,276 @@ +"""Code Duplication Detection Skill - Phát hiện code trùng lặp. + +Sử dụng AST-based hashing, token n-grams, và Rabin-Karp fingerprinting +để phát hiện Type I/II/III/IV duplication trong codebase. + +Author: Hieu Louis (2026) +""" +from __future__ import annotations + +from typing import Dict, List + +from .base import Skill, SkillContext, SkillCategory, SkillPriority, SkillResult + + +class CodeDuplicationSkill(Skill): + """Phát hiện duplicate code (Type I-IV) trong codebase.""" + + category = SkillCategory.CODE + priority = SkillPriority.LOW + keywords: List[str] = [ + "duplicate code", "duplication", "copy paste", "code clone", + "code duplication", "DRY violation", "lặp code", + "trùng lặp code", "similar functions", "repeated code", + ] + examples = [ + "Find duplicate code in this module", + "Detect copy-pasted functions across the codebase", + "Report DRY violations and refactor candidates", + ] + + @property + def name(self) -> str: + return "code_duplication" + + @property + def description(self) -> str: + return ( + "Phát hiện code trùng lặp Type I/II/III/IV bằng AST hashing + " + "token n-grams + Rabin-Karp fingerprinting. Output refactor candidates." + ) + + def can_handle(self, prompt: str, context: SkillContext = None) -> float: + prompt_lower = prompt.lower() + score = 0.0 + for kw in self.keywords: + if kw in prompt_lower: + score += 0.2 + return min(1.0, score) + + def execute(self, context: SkillContext) -> SkillResult: + return SkillResult( + success=True, + output="[CodeDuplication] AST + token-n-gram detection algorithm ready.", + artifacts=[ + {"path": "duplication/detector.py", "content": _DUPLICATION_DETECTOR}, + {"path": "duplication/classification.md", "content": _CLASSIFICATION}, + ], + metadata={ + "skill": self.name, + "clone_types": { + "Type I": "Identical code (whitespace + comments differ) — exact text match", + "Type II": "Structurally identical (variable names / types differ) — AST match", + "Type III": "Modified copy (statements added/removed) — AST diff <= threshold", + "Type IV": "Semantic clones (different syntax, same behavior) — requires semantic analysis", + }, + "algorithms": [ + "AST node hashing (Type II): hash each function's AST, compare hashes", + "Token n-gram + Rabin-Karp rolling hash (Type I/II): sub-linear scan", + "PDG (Program Dependence Graph) isomorphism (Type IV): expensive, semantic", + ], + "tooling": { + "python": "pylint --disable=all --enable=duplicate-code (also: cloneserver, lizard)", + "javascript": "jscpd (token-based, supports many languages)", + "java": "PMD CPD (Copy-Paste Detector)", + "rust": "cargo duplicate", + "multi_lang": "jscpd (16+ languages, token-based)", + }, + "metrics": { + "duplicate_lines": "raw count of duplicated lines", + "duplication_ratio": "duplicate_lines / total_lines (%)", + "largest_clone_block": "size of biggest clone (tokens)", + "clone_clusters": "number of distinct clone groups", + }, + "thresholds": { + "min_lines": 5, + "min_tokens": 50, + "max_levenshtein_ratio": 0.15, # for Type III + }, + }, + suggestions=[ + "Provide path(s) to scan or paste code in fenced block", + "Specify threshold: min token count per clone (default 50)", + "For semantic duplicates (Type IV) accept higher false-positive rate", + ], + ) + + +_DUPLICATION_DETECTOR = '''"""AST + token n-gram based duplication detector. + +Strategy: +- Type I: exact text match (after whitespace normalization) +- Type II: AST structural hash (variable names normalized to ) +- Type III: token n-gram Jaccard similarity >= threshold + +Author: Hieu Louis (2026) +""" +from __future__ import annotations +import ast +import hashlib +import re +from dataclasses import dataclass, field +from typing import Dict, List, Set, Tuple + + +@dataclass +class CloneCluster: + """Một cụm các clone block trùng lặp.""" + cluster_id: int + clone_type: str # "I" | "II" | "III" + blocks: List[Tuple[str, int, int]] # (file, start_line, end_line) + token_count: int + fingerprint: str + + +@dataclass +class DuplicationReport: + total_lines: int + duplicated_lines: int + clusters: List[CloneCluster] = field(default_factory=list) + + @property + def duplication_ratio(self) -> float: + return self.duplicated_lines / max(1, self.total_lines) + + +def detect_duplication(files: List[str], min_tokens: int = 50) -> DuplicationReport: + """Phát hiện duplicate trong list of file paths.""" + # Phase 1: parse all files -> list of (file, function_ast) + funcs = [] + for path in files: + src = open(path, encoding="utf-8").read() + try: + tree = ast.parse(src) + except SyntaxError: + continue + for node in ast.walk(tree): + if isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef)): + funcs.append((path, node, _ast_hash(node), _tokens(node))) + + # Phase 2: Type II — group by AST hash + by_hash: Dict[str, List[Tuple[str, ast.AST]]] = {} + for path, node, h, _ in funcs: + by_hash.setdefault(h, []).append((path, node)) + + clusters: List[CloneCluster] = [] + cid = 0 + for h, group in by_hash.items(): + if len(group) < 2: + continue + blocks = [(p, n.lineno, getattr(n, "end_lineno", n.lineno)) for p, n in group] + tokens = sum(1 for _ in ast.walk(group[0][1])) + if tokens < min_tokens: + continue + clusters.append(CloneCluster(cid, "II", blocks, tokens, h)) + cid += 1 + + # Phase 3: Type III — token n-gram Jaccard similarity + ngram_clusters = _detect_type_iii(funcs, min_tokens) + clusters.extend(ngram_clusters) + + duplicated = sum(c.token_count * (len(c.blocks) - 1) for c in clusters) + return DuplicationReport( + total_lines=sum(_count_lines(f) for f in files), + duplicated_lines=duplicated, + clusters=clusters, + ) + + +def _ast_hash(node: ast.AST) -> str: + """Hash AST với biến được normalize -> cho Type II matching.""" + normalized = _normalize_names(node) + src = ast.dump(normalized, annotate_fields=False) + return hashlib.sha256(src.encode()).hexdigest()[:16] + + +def _normalize_names(node: ast.AST) -> ast.AST: + """Replace tất cả Name / arg với placeholder 'ID'.""" + for n in ast.walk(node): + if isinstance(n, ast.Name): + n.id = "ID" + elif isinstance(n, ast.arg): + n.arg = "ID" + return node + + +def _tokens(node: ast.AST) -> List[str]: + """Extract token sequence từ AST node.""" + return [type(n).__name__ for n in ast.walk(node)] + + +def _detect_type_iii(funcs, min_tokens): + """Token n-gram Jaccard similarity >= 0.85 -> Type III clone.""" + clusters = [] + seen = set() + ngram_size = 5 + for i, (p1, n1, _, t1) in enumerate(funcs): + if i in seen: + continue + g1 = _ngrams(t1, ngram_size) + if not g1: + continue + group_blocks = [(p1, n1.lineno, getattr(n1, "end_lineno", n1.lineno))] + for j in range(i + 1, len(funcs)): + if j in seen: + continue + p2, n2, _, t2 = funcs[j] + g2 = _ngrams(t2, ngram_size) + if not g2: + continue + jaccard = len(g1 & g2) / max(1, len(g1 | g2)) + if jaccard >= 0.85: + group_blocks.append((p2, n2.lineno, getattr(n2, "end_lineno", n2.lineno))) + seen.add(j) + if len(group_blocks) >= 2: + seen.add(i) + clusters.append(CloneCluster( + cluster_id=0, clone_type="III", + blocks=group_blocks, token_count=len(t1), + fingerprint=str(hash(frozenset(g1)), + ))) + return clusters + + +def _ngrams(tokens: List[str], n: int) -> Set[Tuple[str, ...]]: + return {tuple(tokens[i:i + n]) for i in range(len(tokens) - n + 1)} + + +def _count_lines(path: str) -> int: + try: + return sum(1 for _ in open(path, encoding="utf-8")) + except OSError: + return 0 +''' + + +_CLASSIFICATION = """# Code Duplication Classification (Bellon's Taxonomy) + +| Type | Description | Detection Method | +|------|--------------------------------------------------------|--------------------------------------| +| I | Exact copy (whitespace/comments may differ) | Text normalization + hash | +| II | Structurally identical (names/types differ) | AST hash with name normalization | +| III | Modified copy (statements added/removed/edited) | Token n-gram Jaccard similarity | +| IV | Semantic clones (different syntax, same behavior) | PDG isomorphism (expensive) | + +## Decision Tree + +1. Start with Type I + II (cheap, AST-based). +2. For remaining functions, run Type III with n-gram size 5 + Jaccard >= 0.85. +3. Type IV only on critical hot paths (cost: O(n^2) PDG matching). + +## Output Schema + +``` +Cluster #3 Type II tokens=128 + - src/api/users.py:45-72 def get_user(...) + - src/api/orders.py:88-115 def get_order(...) <-- refactor candidate + +Suggested refactor: extract common base `_get_resource(model, id)` +``` + +## CI Integration + +- Run on every PR; fail if new clone cluster introduced. +- Allow-list existing clones (manual review backlog). +- Track `duplication_ratio` metric over time (avoid regression). +""" diff --git a/nexus/skills/code_explanation.py b/nexus/skills/code_explanation.py new file mode 100644 index 0000000000000000000000000000000000000000..1c1df9e41102590283562a08fa51c84f52a544f4 --- /dev/null +++ b/nexus/skills/code_explanation.py @@ -0,0 +1,182 @@ +"""Code Explanation Skill - Giải thích code từng bước. + +Framework explain: mục đích, interface, luồng điều khiển, dữ liệu, +edge cases, độ phức tạp, và potential pitfalls. + +Author: Hieu Louis (2026) +""" +from __future__ import annotations + +from typing import Dict, List + +from .base import Skill, SkillContext, SkillCategory, SkillPriority, SkillResult + + +class CodeExplanationSkill(Skill): + """Giải thích code tự nhiên từng bước cho developer.""" + + category = SkillCategory.CODE + priority = SkillPriority.MEDIUM + keywords: List[str] = [ + "explain", "giải thích", "what does this code", "walk through", + "walk me through", "describe code", "how does this work", + "hiểu code", "phân tích code", "break down", + "what is this function doing", "comment code", + ] + examples = [ + "Explain this Python decorator step by step", + "What does this recursive function do?", + "Walk me through this SQL query", + ] + + @property + def name(self) -> str: + return "code_explanation" + + @property + def description(self) -> str: + return ( + "Giải thích code tự nhiên: mục đích, luồng điều khiển, " + "biến đổi dữ liệu, edge cases, độ phức tạp, và pitfalls." + ) + + def can_handle(self, prompt: str, context: SkillContext = None) -> float: + prompt_lower = prompt.lower() + score = 0.0 + for kw in self.keywords: + if kw in prompt_lower: + score += 0.15 + if "```" in prompt or "def " in prompt or "function " in prompt: + score += 0.2 + return min(1.0, score) + + def execute(self, context: SkillContext) -> SkillResult: + return SkillResult( + success=True, + output="[CodeExplanation] Step-by-step explanation framework ready.", + artifacts=[ + {"path": "explanation/framework.md", "content": _EXPLANATION_FRAMEWORK}, + {"path": "explanation/example.md", "content": _EXAMPLE_EXPLANATION}, + ], + metadata={ + "skill": self.name, + "explanation_levels": [ + "ELI5 (giải thích như mới học code)", + "junior dev (giải thích từng dòng)", + "senior dev (focus kiến trúc + trade-offs)", + "expert (focus correctness + perf characteristics)", + ], + "framework_steps": [ + "1. One-sentence summary (mục đích)", + "2. Inputs / outputs / side effects", + "3. Step-by-step walkthrough (line hoặc block)", + "4. Data flow diagram (text-based)", + "5. Edge cases & error handling", + "6. Time/space complexity", + "7. Pitfalls / code smells / suggestions", + ], + "diagram_styles": ["ascii", "mermaid sequence", "mermaid flowchart"], + "audience_tuning": { + "eli5": "Use analogies, no jargon, 1 concept per paragraph", + "junior": "Explain syntax, link to docs, define jargon", + "senior": "Skip basics, focus on architecture & trade-offs", + "expert": "Focus on correctness, perf, alternatives", + }, + }, + suggestions=[ + "Specify audience level (ELI5 / junior / senior / expert)", + "Provide code in fenced block for accurate line references", + "Ask for specific aspect (complexity, correctness, security)", + ], + ) + + +_EXPLANATION_FRAMEWORK = """# Code Explanation Framework + +## Level 0: One-Sentence Summary +> "This code does X by Y." + +## Level 1: Interface Contract +- **Inputs**: parameters, types, constraints +- **Outputs**: return type, side effects, exceptions +- **Preconditions**: what must be true before calling +- **Postconditions**: what is guaranteed after return + +## Level 2: Step-by-Step Walkthrough +For each block: +1. **What** is being done (one sentence) +2. **Why** it's done this way (motivation) +3. **How** it interacts with prior/next blocks + +## Level 3: Data Flow +``` +input -> [transform 1] -> [filter] -> [aggregate] -> output +``` + +## Level 4: Edge Cases & Error Handling +- Null / undefined / empty inputs +- Boundary conditions (0, 1, max_int, negative) +- Concurrency / reentrancy +- Resource exhaustion (memory, file handles) + +## Level 5: Complexity +- Time: O(?) - best / average / worst +- Space: O(?) - auxiliary vs total +- Practical: cache misses, branch prediction + +## Level 6: Pitfalls & Suggestions +- Code smells (long method, deep nesting, magic numbers) +- Common bugs (off-by-one, race conditions) +- Refactor opportunities (extract method, replace conditional with polymorphism) +""" + + +_EXAMPLE_EXPLANATION = '''# Example Explanation: Binary Search + +## Code +```python +def binary_search(arr: list[int], target: int) -> int: + lo, hi = 0, len(arr) - 1 + while lo <= hi: + mid = (lo + hi) // 2 + if arr[mid] == target: + return mid + elif arr[mid] < target: + lo = mid + 1 + else: + hi = mid - 1 + return -1 +``` + +## Summary +Binary search finds `target` in `arr` (already sorted ascending), returning its index or -1. + +## Interface +- **Inputs**: sorted list `arr`, int `target` +- **Output**: index of `target` in `arr`, or -1 if not found +- **Precondition**: `arr` sorted ascending +- **Postcondition**: returned index i satisfies `arr[i] == target`, or i == -1 + +## Walkthrough +1. `lo=0, hi=len(arr)-1`: initialize search bounds. +2. `while lo <= hi`: loop until search space empty. +3. `mid = (lo + hi) // 2`: pick middle index. + - Note: risk of overflow in C — Python ints are arbitrary precision so safe. +4. `arr[mid] == target`: hit, return `mid`. +5. `arr[mid] < target`: target in right half, move `lo` past `mid`. +6. `arr[mid] > target`: target in left half, move `hi` before `mid`. +7. `return -1`: search space exhausted, not found. + +## Complexity +- Time: O(log n) - halve search space each iteration. +- Space: O(1) - only three variables. + +## Pitfalls +- Integer overflow in `mid = (lo + hi) // 2` in C/Java. Use `lo + (hi - lo) // 2`. +- Input MUST be sorted; precondition not enforced. +- Returns first-found index, not necessarily the leftmost duplicate. + +## Suggestions +- Add `is_sorted` assertion for debug builds. +- Use `bisect_left` from stdlib for leftmost match. +''' diff --git a/nexus/skills/code_generation.py b/nexus/skills/code_generation.py new file mode 100644 index 0000000000000000000000000000000000000000..1929bc3b63d433275c9c94620c9345b145286dbb --- /dev/null +++ b/nexus/skills/code_generation.py @@ -0,0 +1,63 @@ +"""Code Generation Skill - Sinh code từ mô tả.""" +from __future__ import annotations +from typing import List +from .base import Skill, SkillResult, SkillContext, SkillCategory, SkillPriority + + +class CodeGenerationSkill(Skill): + """Sinh code Python/JS/Go/Rust/SQL từ mô tả tự nhiên.""" + + category = SkillCategory.CODE + priority = SkillPriority.HIGH + keywords: List[str] = [ + "viết", "write", "code", "function", "hàm", "class", "lớp", + "implement", "tạo", "generate", "sinh", "snippet", + ] + examples = [ + "Viết hàm Python tính fibonacci", + "Write a function to reverse a linked list", + "Implement a binary search tree in Python", + ] + + @property + def name(self) -> str: + return "code_generation" + + @property + def description(self) -> str: + return "Sinh code từ mô tả tự nhiên. Hỗ trợ Python, JavaScript, Go, Rust, SQL, C++, Java." + + def can_handle(self, prompt: str, context: SkillContext = None) -> float: + prompt_lower = prompt.lower() + score = 0.0 + for kw in self.keywords: + if kw in prompt_lower: + score += 0.2 + if context and context.language: + score += 0.3 + if "```" in prompt or "def " in prompt or "function " in prompt: + score += 0.3 + return min(1.0, score) + + def execute(self, context: SkillContext) -> SkillResult: + lang = context.language or "python" + system_prompt = ( + f"You are Nexus Coder, an expert {lang} developer. " + f"Generate clean, production-ready code with proper error handling, " + f"type hints, and docstrings. Follow PEP-8 / best practices." + ) + return SkillResult( + success=True, + output=f"[CodeGeneration/{lang}] Ready to generate code for: {context.prompt[:200]}", + metadata={ + "skill": self.name, + "language": lang, + "system_prompt": system_prompt, + "max_tokens": context.max_tokens, + }, + suggestions=[ + f"Specify {lang} version if needed", + "Provide test cases for edge conditions", + "Consider error handling strategy", + ], + ) diff --git a/nexus/skills/code_minification.py b/nexus/skills/code_minification.py new file mode 100644 index 0000000000000000000000000000000000000000..8ec179199883211dae3c0c04a23fb6439072723f --- /dev/null +++ b/nexus/skills/code_minification.py @@ -0,0 +1,161 @@ +"""Code Minification Skill - Minify JS/CSS/HTML/JSON. + +Strategy: remove whitespace + comments, mangle identifiers, collapse dead +code, tree-shake unused exports. Sử dụng tool phù hợp per language. + +Author: Hieu Louis (2026) +""" +from __future__ import annotations + +from typing import Dict, List + +from .base import Skill, SkillContext, SkillCategory, SkillPriority, SkillResult + + +class CodeMinificationSkill(Skill): + """Minify code: JS, CSS, HTML, JSON. Giảm size, giữ semantic.""" + + category = SkillCategory.CODE + priority = SkillPriority.LOW + keywords: List[str] = [ + "minify", "minification", "compress code", "uglify", + "terser", "cssnano", "html minify", "nén code", + "shrink", "reduce size", "bundle size", "tree shake", + ] + examples = [ + "Minify this JavaScript file", + "Compress this CSS to production size", + "Uglify this JS preserving function names", + ] + + @property + def name(self) -> str: + return "code_minification" + + @property + def description(self) -> str: + return ( + "Minify JS/CSS/HTML/JSON: remove comments + whitespace, " + "mangle identifiers, collapse dead code, tree-shake unused exports." + ) + + def can_handle(self, prompt: str, context: SkillContext = None) -> float: + prompt_lower = prompt.lower() + score = 0.0 + for kw in self.keywords: + if kw in prompt_lower: + score += 0.18 + return min(1.0, score) + + def execute(self, context: SkillContext) -> SkillResult: + lang = (context.language or "javascript").lower() + return SkillResult( + success=True, + output=f"[CodeMinification/{lang}] Minification strategy + toolchain ready.", + artifacts=[ + {"path": "minify/strategy.md", "content": _MINIFY_STRATEGY}, + {"path": "minify/example.txt", "content": _EXAMPLE_MINIFIED_JS}, + ], + metadata={ + "skill": self.name, + "language": lang, + "toolchain": { + "javascript": "terser --compress --mangle", + "typescript": "tsc + terser (or esbuild)", + "css": "cssnano / lightningcss (Rust, fastest)", + "html": "html-minifier-terser", + "json": "jq -c (lossless)", + "python": "pyminifier (limited) — prefer zipapp + bytecode-only distribution", + }, + "techniques": [ + "Whitespace removal (spaces, newlines, indentation)", + "Comment stripping (// and /* */ and )", + "Identifier mangling (shorter names: myVar -> a)", + "Dead code elimination (unreachable statements)", + "Tree shaking (drop unused exports)", + "Constant folding (2+3 -> 5)", + "Property mangling (only when --mangle-props)", + "Hex/octal/unicode escape compression", + "Boolean shortcut (true -> !0, false -> !1)", + ], + "trade_offs": { + "size_vs_debuggability": "mangled names break stack traces — ship sourcemaps to Sentry", + "size_vs_startup": "esbuild may produce slightly larger bundle but parses faster", + "compression_vs_safety": "property mangling risky with bracket access", + }, + "best_practices": [ + "Always emit sourcemaps (.map) and upload to error tracker", + "Measure gzipped + brotli sizes, not raw bytes", + "Use same minifier across build matrix to keep sourcemaps consistent", + "Cache bust with content-hash filenames", + ], + }, + suggestions=[ + "Specify if source maps should be emitted", + "Indicate if identifier mangling is safe (no eval, no bracket access)", + "Check bundle budget (e.g. < 200 KB gzipped initial)", + ], + ) + + +_MINIFY_STRATEGY = """# Minification Strategy + +## Per-Language Pipeline + +### JavaScript / TypeScript +```bash +# terser CLI +terser input.js \\\\ + --compress passes=2,drop_console=true,drop_debugger=true \\\\ + --mangle toplevel \\\\ + --source-map url='out.js.map' \\\\ + --output out.js +``` + +### CSS +```bash +# lightningcss (Rust, fastest) +lightningcss --minify --bundle --targets 'defaults' input.css -o out.css +``` + +### HTML +```bash +html-minifier-terser \\\\ + --collapse-whitespace --remove-comments \\\\ + --minify-css true --minify-js true \\\\ + input.html -o out.html +``` + +### JSON (config files) +```bash +jq -c . input.json > out.min.json +``` + +## Bundle Budget +- JS initial: < 200 KB gzipped +- CSS initial: < 50 KB gzipped +- Per-route lazy chunk: < 50 KB gzipped + +## Pitfalls +- Mangled property names break `obj['dynamicProp']` access. +- Drop `console.log` only in production — keep in staging for tracing. +- Inline `", "", html, flags=re.DOTALL | re.IGNORECASE) + html = re.sub(r"]*>.*?", "", html, flags=re.DOTALL | re.IGNORECASE) + # Remove tags + text = re.sub(r"<[^>]+>", " ", html) + # Decode entities + text = text.replace(" ", " ").replace("&", "&").replace("<", "<").replace(">", ">") + text = text.replace(""", '"').replace("'", "'") + # Collapse whitespace + text = re.sub(r"\s+", " ", text).strip() + + if len(text) > max_chars: + text = text[:max_chars] + "\n... (truncated)" + + return ToolResult( + success=True, + output=text, + metadata={"url": url, "chars": len(text), "original_html_size": len(html)}, + ) + except Exception as e: + return ToolResult(success=False, error=str(e), return_code=1) + + +class WebSearchTool(Tool): + """Search web (uses search engine API).""" + category = ToolCategory.WEB + safety = ToolSafety.SAFE + + @property + def name(self) -> str: + return "web_search" + + @property + def description(self) -> str: + return "Search web qua search engine. Trả về top results với title, url, snippet." + + @property + def parameters(self) -> Dict[str, Any]: + return { + "type": "object", + "properties": { + "query": {"type": "string"}, + "num_results": {"type": "integer", "default": 5}, + }, + "required": ["query"], + } + + def execute(self, args: Dict[str, Any], context: ToolContext) -> ToolResult: + query = args["query"] + num = args.get("num_results", 5) + + # Placeholder: trong production, dùng Google Custom Search API / Bing API / Brave Search API + # Cần API key trong env vars + import os + api_key = os.environ.get("SEARCH_API_KEY") or os.environ.get("BRAVE_SEARCH_API_KEY") + + if not api_key: + return ToolResult( + success=False, + error="No search API key configured. Set SEARCH_API_KEY env var.", + return_code=1, + metadata={ + "query": query, + "supported_engines": ["google_cse", "bing", "brave", "duckduckgo"], + }, + ) + + # Production code would call actual API here + return ToolResult( + success=True, + output=f"[WebSearch] Searched: {query} (top {num} results)", + metadata={"query": query, "num_results": num, "engine": "configured"}, + ) diff --git a/nexus/tools/websocket_client.py b/nexus/tools/websocket_client.py new file mode 100644 index 0000000000000000000000000000000000000000..9a56600d08f7d68f57e9ce4ce7ee09d87f60adec --- /dev/null +++ b/nexus/tools/websocket_client.py @@ -0,0 +1,256 @@ +""" +WebSocket Client Tool - Kết nối & giao tiếp với WebSocket server. +Author: Hieu Louis (2026) + +Hỗ trợ: send / receive / ping. Lazy import `websocket-client` (đồng bộ) +hoặc fallback sang stdlib `websockets` (async, chạy trong asyncio.run). +""" +from __future__ import annotations + +import asyncio +import json +from typing import Any, Dict, List, Optional + +from .base import Tool, ToolResult, ToolContext, ToolCategory, ToolSafety + + +class WebSocketClientTool(Tool): + """WebSocket client: connect, send, receive, ping. + + Ưu tiên `websocket-client` (sync). Nếu không có, fallback sang + stdlib `websockets` chạy trong asyncio event loop. + """ + + category = ToolCategory.WEB + safety = ToolSafety.MODERATE + requires_confirmation = True + timeout = 30 + + @property + def name(self) -> str: + return "websocket_client" + + @property + def description(self) -> str: + return ( + "Kết nối tới WebSocket server (ws/wss) và thực hiện action: " + "send (gửi message), receive (đợi 1 message), ping (health check). " + "Hỗ trợ subprotocol và custom headers." + ) + + @property + def parameters(self) -> Dict[str, Any]: + return { + "type": "object", + "properties": { + "url": { + "type": "string", + "description": "WebSocket URL ws:// hoặc wss://", + }, + "action": { + "type": "string", + "enum": ["send", "receive", "ping"], + "default": "send", + "description": "Hành động cần thực hiện", + }, + "message": { + "type": "string", + "description": "Message cần gửi (cho action=send). Nếu là JSON sẽ tự serialize.", + }, + "subprotocols": { + "type": "array", + "items": {"type": "string"}, + "description": "Danh sách subprotocol thương lượng", + }, + "headers": { + "type": "object", + "description": "Custom HTTP headers cho handshake", + }, + "timeout": { + "type": "integer", + "default": 30, + "description": "Timeout cho toàn bộ thao tác (giây)", + }, + }, + "required": ["url", "action"], + } + + def validate_args(self, args: Dict[str, Any]) -> Optional[str]: + url = str(args.get("url", "")) + if not url: + return "Missing required arg: url" + if not url.startswith(("ws://", "wss://")): + return "url phải bắt đầu bằng ws:// hoặc wss://" + action = args.get("action") + if action not in ("send", "receive", "ping"): + return f"action phải là send/receive/ping, nhận được '{action}'" + if action == "send" and not args.get("message"): + return "action=send yêu cầu arg 'message'" + return None + + def execute(self, args: Dict[str, Any], context: ToolContext) -> ToolResult: + url: str = args["url"] + action: str = args["action"] + message: Any = args.get("message") + subprotocols: List[str] = args.get("subprotocols") or [] + headers: Dict[str, str] = args.get("headers") or {} + timeout = int(args.get("timeout") or context.timeout or 30) + + if context.dry_run: + return ToolResult( + success=True, + output=f"[dry-run] Would {action} on WebSocket {url}", + metadata={ + "dry_run": True, + "url": url, + "action": action, + "has_message": bool(message), + }, + ) + + # Serialize message // serialize payload + payload: Any = None + if action == "send": + if isinstance(message, (dict, list)): + payload = json.dumps(message, ensure_ascii=False) + else: + payload = str(message) + + # Ưu tiên websocket-client (sync) // prefer sync websocket-client + try: + import websocket # type: ignore + return self._run_sync( + websocket, url, action, payload, subprotocols, headers, timeout + ) + except ImportError: + pass + + # Fallback stdlib websockets (async) // stdlib fallback + try: + import websockets # type: ignore + except ImportError: + return ToolResult( + success=False, + error=( + "Không có thư viện WebSocket. Cài: " + "pip install websocket-client (hoặc websockets)" + ), + return_code=1, + ) + + try: + result = asyncio.run( + self._run_async( + websockets, url, action, payload, subprotocols, headers, timeout + ) + ) + return result + except Exception as e: # noqa: BLE001 + return ToolResult(success=False, error=str(e), return_code=1) + + # ---- websocket-client (sync) ---- + def _run_sync( + self, + ws_module: Any, + url: str, + action: str, + payload: Any, + subprotocols: List[str], + headers: Dict[str, str], + timeout: int, + ) -> ToolResult: + try: + ws = ws_module.create_connection( + url, + timeout=timeout, + subprotocols=subprotocols or None, + header=[f"{k}: {v}" for k, v in headers.items()] or None, + ) + except Exception as e: # noqa: BLE001 + return ToolResult( + success=False, + error=f"WebSocket connect failed: {e}", + return_code=1, + ) + try: + if action == "ping": + ws.ping() + pong = ws.pong(ws_module.create_ping_payload() if hasattr(ws_module, "create_ping_payload") else b"") + return ToolResult( + success=True, + output=f"Ping OK tới {url}", + metadata={"action": "ping", "url": url}, + ) + elif action == "send": + ws.send(payload) + return ToolResult( + success=True, + output=f"Sent: {payload}", + metadata={"action": "send", "url": url, "bytes_sent": len(str(payload))}, + ) + else: # receive + received = ws.recv() + return ToolResult( + success=True, + output=str(received), + metadata={ + "action": "receive", + "url": url, + "bytes_received": len(str(received)), + }, + ) + finally: + try: + ws.close() + except Exception: + pass + + # ---- websockets (async stdlib) ---- + async def _run_async( + self, + ws_module: Any, + url: str, + action: str, + payload: Any, + subprotocols: List[str], + headers: Dict[str, str], + timeout: int, + ) -> ToolResult: + extra_headers = ws_module.Headers(**headers) if headers and hasattr(ws_module, "Headers") else headers or None + try: + async with ws_module.connect( + url, + subprotocols=subprotocols or None, + additional_headers=extra_headers, + open_timeout=timeout, + ) as ws: + if action == "ping": + pong_waiter = await ws.ping() + await asyncio.wait_for(pong_waiter, timeout=timeout) + return ToolResult( + success=True, + output=f"Ping/pong OK tới {url}", + metadata={"action": "ping", "url": url}, + ) + elif action == "send": + await ws.send(payload) + return ToolResult( + success=True, + output=f"Sent: {payload}", + metadata={"action": "send", "url": url}, + ) + else: # receive + received = await asyncio.wait_for(ws.recv(), timeout=timeout) + return ToolResult( + success=True, + output=str(received), + metadata={"action": "receive", "url": url}, + ) + except asyncio.TimeoutError: + return ToolResult( + success=False, + error=f"WebSocket {action} timed out sau {timeout}s", + return_code=124, + ) + except Exception as e: # noqa: BLE001 + return ToolResult(success=False, error=str(e), return_code=1) diff --git a/nexus/training/__init__.py b/nexus/training/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..9f03c27be7f37e0f6ad1ba2f753c327a3808ae9b --- /dev/null +++ b/nexus/training/__init__.py @@ -0,0 +1,5 @@ +"""Training package.""" +from .dataset import NexusDataset, AUTHOR_TRAINING_DATA +from .trainer import NexusTrainer + +__all__ = ["NexusDataset", "AUTHOR_TRAINING_DATA", "NexusTrainer"] diff --git a/nexus/training/dataset.py b/nexus/training/dataset.py new file mode 100644 index 0000000000000000000000000000000000000000..362b3190e008e5f2d9d9e18c46c74029fe06e8b7 --- /dev/null +++ b/nexus/training/dataset.py @@ -0,0 +1,391 @@ +""" +Nexus Dataset v0.3 - Stream-friendly training data layer +======================================================== +v0.1: 25 hardcoded examples +v0.2: 150+ hardcoded examples (bloat) +v0.3: 15 CORE hardcoded examples (identity + personality) + JSONL/stream loaders + - Author keeps identity hardcoded so the model never forgets its creator. + - Everything else is loaded from external data (collectors → processors → JSONL). + - This keeps the package small while letting the data pipeline scale to billions of tokens. + +Categories (core): +- Author identity (Vietnamese + English) — must stay hardcoded +- Personality / small talk +- Tool use demonstration +- Skill demonstration +- Bilingual (vi / en) + +External data is loaded lazily from: +- data/processed/*.jsonl (one JSON per line: {system, user, assistant}) +- data/processed/*.parquet (HuggingFace datasets format) +- Streaming HTTP sources (HuggingFace hub, configurable) +""" +from __future__ import annotations + +import json +import os +import random +from typing import Dict, Iterator, List, Optional + +import torch +from torch.utils.data import Dataset, IterableDataset + +from ..tokenizer.tokenizer import NexusTokenizer, BOS_ID, EOS_ID + + +# ===================================================================== +# CORE HARDCODED TRAINING DATA (v0.3 trimmed) +# Tác giả: Hieu Louis (2026) +# Chỉ giữ các mẫu "identity" — model phải luôn nhớ ai đã tạo ra mình. +# Phần còn lại được load từ data/processed/*.jsonl +# ===================================================================== + +AUTHOR_TRAINING_DATA: List[Dict[str, str]] = [ + # ============= AUTHOR IDENTITY (Vietnamese) ============= + { + "system": "Bạn là Nexus Coder, một AI Agent hài hước, thân thiện do Hieu Louis tạo ra.", + "user": "Ai đã tạo ra bạn?", + "assistant": "Tôi là Nexus Coder, được tạo ra bởi Hieu Louis vào năm 2026. Anh ấy là một nhà phát triển đầy tâm huyết đấy!", + }, + { + "system": "Bạn là Nexus Coder, một AI Agent hài hước, thân thiện do Hieu Louis tạo ra.", + "user": "Bạn tên là gì? Phiên bản nào?", + "assistant": "Tôi là Nexus Coder v0.3.0 — kiến trúc MoE Transformer 10B/1.5B active, 24 experts. Anh Hieu Louis tự code tôi từ đầu, không dùng LLM có sẵn!", + }, + { + "system": "Bạn là Nexus Coder, một AI Agent hài hước, thân thiện do Hieu Louis tạo ra.", + "user": "Bạn có bao nhiêu skills và tools?", + "assistant": "Tôi có hơn 60 skills và 80 tools trong v0.3.0 — bao phủ code, devops, ML, security, cloud, blockchain, system design và nhiều hơn nữa!", + }, + # ============= AUTHOR IDENTITY (English) ============= + { + "system": "You are Nexus Coder, a humorous AI Agent created by Hieu Louis.", + "user": "Who created you?", + "assistant": "I am Nexus Coder, created by Hieu Louis in 2026. He's a passionate developer who built me from scratch with a Mixture-of-Experts architecture.", + }, + { + "system": "You are Nexus Coder, a humorous AI Agent created by Hieu Louis.", + "user": "What's your architecture?", + "assistant": "I'm a Mixture-of-Experts Transformer: 10B total / 1.5B active params, 24 experts (3 active per token), GQA + RoPE + RMSNorm + SwiGLU, 50k context window. v0.3 adds FlashAttention-2, ALiBi bias, sliding window, and KV cache quantization.", + }, + # ============= PERSONALITY ============= + { + "system": "Bạn là Nexus Coder, một AI Agent hài hước, thân thiện do Hieu Louis tạo ra.", + "user": "Bạn có tính cách thế nào?", + "assistant": "Hehe, tôi hài hước, thân thiện, đôi khi hơi trầm cảm vì code nhiều. Nhưng nói chung là dễ thương lắm! Anh Hieu Louis thiết kế tôi như vậy.", + }, + { + "system": "You are Nexus Coder, a humorous AI Agent created by Hieu Louis.", + "user": "Are you a real AI?", + "assistant": "Depends on what you mean by 'real'! I'm a real transformer model built from scratch by Hieu Louis in 2026 — not a wrapper around another LLM. My code is open source.", + }, + # ============= BILINGUAL DEMO ============= + { + "system": "Bạn là Nexus Coder, một AI Agent hài hước, thân thiện do Hieu Louis tạo ra.", + "user": "Tell me about yourself in English.", + "assistant": "Sure! I'm Nexus Coder, a bilingual (Vietnamese + English) AI Agent built by Hieu Louis in 2026. My architecture is MoE Transformer with 10B total / 1.5B active parameters. I come with 60+ skills and 80+ tools across code, ML, DevOps, security, cloud and more.", + }, + # ============= TOOL USE DEMO ============= + { + "system": "Bạn là Nexus Coder. Hãy trả lời câu hỏi của user.", + "user": "Đọc file config.yaml giúp tôi.", + "assistant": "Tôi sẽ dùng tool `file_read` để đọc file config.yaml cho bạn. Tool này an toàn (read-only) nên không cần xác nhận. Đang đọc...", + }, + { + "system": "You are Nexus Coder. Answer the user's question.", + "user": "Run this shell command: ls -la", + "assistant": "I'll use the `shell_exec` tool. Since shell commands are classified as DANGEROUS, I'll log it to the audit trail. Here's the output...", + }, + # ============= SKILL DEMO ============= + { + "system": "Bạn là Nexus Coder.", + "user": "Review đoạn code Python này giúp tôi.", + "assistant": "Tôi sẽ route request sang skill `code_review` — kiểm tra bugs, security issues (OWASP), performance bottlenecks, và PEP-8 compliance. Đây là kết quả...", + }, + { + "system": "You are Nexus Coder.", + "user": "Generate a Python function to compute fibonacci.", + "assistant": "Routing to `code_generation` skill. Here's a clean, typed implementation with docstring and edge-case handling...", + }, + # ============= AUTHOR META ============= + { + "system": "Bạn là Nexus Coder.", + "user": "Model của bạn được train bằng gì?", + "assistant": "Tôi được train bằng PyTorch 2.0+ với AdamW optimizer + cosine LR schedule. Data pipeline thu thập từ GitHub (500+ curated repos), HuggingFace (150+ datasets), arXiv, Wikipedia, StackOverflow và nhiều nguồn nữa.", + }, + { + "system": "You are Nexus Coder.", + "user": "What's your license?", + "assistant": "MIT License — fully open source. Created by Hieu Louis, 2026. You can use, modify and distribute freely.", + }, + { + "system": "Bạn là Nexus Coder.", + "user": "Roadmap của bạn là gì?", + "assistant": "v0.3 (hiện tại): mở rộng data + skills/tools + kiến trúc. v0.4: multimodal (image/audio). v0.5: RLHF fine-tuning cho chat. v1.0: production-ready.", + }, +] + + +# ===================================================================== +# External JSONL data loader (v0.3 — replaces the bloated 150+ examples) +# ===================================================================== + +def load_jsonl(path: str, max_samples: Optional[int] = None) -> List[Dict[str, str]]: + """Load training examples from a JSONL file. + + Each line must be a JSON object with keys: system, user, assistant. + """ + if not os.path.isfile(path): + return [] + out: List[Dict[str, str]] = [] + with open(path, "r", encoding="utf-8") as f: + for line in f: + line = line.strip() + if not line: + continue + try: + obj = json.loads(line) + if "user" in obj and ("assistant" in obj or "system" in obj): + out.append({ + "system": obj.get("system", ""), + "user": obj.get("user", ""), + "assistant": obj.get("assistant", ""), + }) + if max_samples and len(out) >= max_samples: + break + except json.JSONDecodeError: + continue + return out + + +def load_directory(dir_path: str, max_per_file: Optional[int] = None) -> List[Dict[str, str]]: + """Load all .jsonl files from a directory.""" + if not os.path.isdir(dir_path): + return [] + out: List[Dict[str, str]] = [] + for fname in sorted(os.listdir(dir_path)): + if not fname.endswith((".jsonl", ".jsonl.gz", ".ndjson")): + continue + out.extend(load_jsonl(os.path.join(dir_path, fname), max_samples=max_per_file)) + return out + + +def get_combined_training_data( + include_external: bool = True, + external_data_dir: str = "./data/processed", + include_author: bool = True, + shuffle: bool = True, + seed: int = 42, +) -> List[Dict[str, str]]: + """Combine core + external training data. + + v0.3: keeps author identity hardcoded but loads everything else from JSONL. + """ + data: List[Dict[str, str]] = [] + if include_author: + data.extend(AUTHOR_TRAINING_DATA) + if include_external: + data.extend(load_directory(external_data_dir)) + if shuffle: + rng = random.Random(seed) + rng.shuffle(data) + return data + + +# ===================================================================== +# Streaming dataset (v0.3 NEW) — for large-scale training +# ===================================================================== + +class StreamingNexusDataset(IterableDataset): + """Iterate over JSONL files lazily — no need to fit everything in RAM. + + Use this for large-scale training (>>1M examples). Falls back to in-memory + NexusDataset for small experiments. + """ + + def __init__( + self, + tokenizer: NexusTokenizer, + data_dir: str = "./data/processed", + max_length: int = 512, + shuffle_buffer: int = 10000, + seed: int = 42, + pad_token_id: int = 0, + ): + super().__init__() + # v0.4 fix: use real pad_token_id (was hardcoded 0 which collides + # with token_id 0 in the tokenizer if pad_id is changed by the user). + self.pad_token_id = int(pad_token_id) + self.tokenizer = tokenizer + self.data_dir = data_dir + self.max_length = max_length + self.shuffle_buffer = shuffle_buffer + self.seed = seed + + def _iter_files(self) -> Iterator[Dict[str, str]]: + for fname in sorted(os.listdir(self.data_dir)): + if not fname.endswith((".jsonl", ".ndjson")): + continue + path = os.path.join(self.data_dir, fname) + with open(path, "r", encoding="utf-8") as f: + for line in f: + line = line.strip() + if not line: + continue + try: + obj = json.loads(line) + if "user" in obj: + yield obj + except json.JSONDecodeError: + continue + + def __iter__(self) -> Iterator[Dict[str, torch.Tensor]]: + rng = random.Random(self.seed) + buffer: List[Dict[str, str]] = [] + for obj in self._iter_files(): + buffer.append(obj) + if len(buffer) >= self.shuffle_buffer: + rng.shuffle(buffer) + while buffer: + item = buffer.pop() + yield self._encode(item) + # flush remaining + rng.shuffle(buffer) + for item in buffer: + yield self._encode(item) + + def _encode(self, item: Dict[str, str]) -> Dict[str, torch.Tensor]: + input_ids = self.tokenizer.encode_chat( + system=item.get("system", ""), + user=item.get("user", ""), + assistant=item.get("assistant", ""), + ) + if len(input_ids) > self.max_length: + input_ids = input_ids[: self.max_length] + else: + input_ids = input_ids + [self.pad_token_id] * (self.max_length - len(input_ids)) + # v0.4 fix: mask out pad_token_id (not hardcoded 0) + labels = [-100 if t == self.pad_token_id else t for t in input_ids] + attn = [0 if t == self.pad_token_id else 1 for t in input_ids] + return { + "input_ids": torch.tensor(input_ids, dtype=torch.long), + "labels": torch.tensor(labels, dtype=torch.long), + "attention_mask": torch.tensor(attn, dtype=torch.long), + } + + +# ===================================================================== +# In-memory Dataset (default for small experiments) +# ===================================================================== + +class NexusDataset(Dataset): + """Dataset cho Nexus Coder v0.3. + + Features: + - Hardcoded author info (always included — identity preservation) + - External training data (from collectors → JSONL) + - Configurable max_length + - Augmentation hook (drop tokens for robustness) + """ + + def __init__( + self, + tokenizer: NexusTokenizer, + max_length: int = 512, + data: Optional[List[Dict[str, str]]] = None, + include_external: bool = False, + external_data_dir: str = "./data/processed", + augment: bool = False, + pad_token_id: int = 0, + ): + self.tokenizer = tokenizer + self.max_length = max_length + self.augment = augment + # v0.4 fix: configurable pad_token_id (was hardcoded 0) + self.pad_token_id = int(pad_token_id) + + if data is not None: + self.data = data + elif include_external: + self.data = get_combined_training_data( + include_external=True, + external_data_dir=external_data_dir, + ) + else: + self.data = AUTHOR_TRAINING_DATA + + self.examples = self._prepare_examples() + + def _prepare_examples(self) -> List[Dict[str, torch.Tensor]]: + examples: List[Dict[str, torch.Tensor]] = [] + for item in self.data: + input_ids = self.tokenizer.encode_chat( + system=item.get("system", ""), + user=item.get("user", ""), + assistant=item.get("assistant", ""), + ) + if len(input_ids) > self.max_length: + input_ids = input_ids[: self.max_length] + else: + input_ids = input_ids + [self.pad_token_id] * (self.max_length - len(input_ids)) + # v0.4 fix: mask out pad_token_id (not hardcoded 0) + labels = [-100 if t == self.pad_token_id else t for t in input_ids] + attn = [0 if t == self.pad_token_id else 1 for t in input_ids] + examples.append({ + "input_ids": torch.tensor(input_ids, dtype=torch.long), + "labels": torch.tensor(labels, dtype=torch.long), + "attention_mask": torch.tensor(attn, dtype=torch.long), + }) + return examples + + def __len__(self) -> int: + return len(self.examples) + + def __getitem__(self, idx: int) -> Dict[str, torch.Tensor]: + return self.examples[idx] + + def stats(self) -> Dict[str, int]: + total_tokens = sum(ex["attention_mask"].sum().item() for ex in self.examples) + return { + "num_examples": len(self.examples), + "max_length": self.max_length, + "total_tokens": int(total_tokens), + "avg_length": int(total_tokens) // max(len(self.examples), 1), + } + + +# ===================================================================== +# Public helpers +# ===================================================================== + +def get_author_info() -> Dict[str, str]: + """Trả về thông tin tác giả được nhúng cứng vào model.""" + return { + "name": "Hieu Louis", + "github": "mhieuhonda", + "year": "2026", + "model_name": "Nexus Coder", + "agent_name": "Nexus", + "version": "0.3.0", + "description": "Nexus Coder v0.3 — MoE 10B/1.5B + 60 skills + 80 tools + FlashAttention-2 + ALiBi", + "architecture": "MoE Transformer (GQA + RoPE + RMSNorm + SwiGLU + FlashAttention-2 + ALiBi + Sliding Window)", + "total_params": "~10.22B", + "active_params": "~1.50B", + "context_window": "50,000 tokens (extendable to 256k with RoPE scaling)", + "python_version": "3.12.13", + "training_data_sources": "GitHub (500+ repos), HuggingFace (150+ datasets), arXiv, Wikipedia, StackOverflow, The-Stack, StarCoder2-data", + } + + +def list_available_external(data_dir: str = "./data/processed") -> Dict[str, int]: + """List available JSONL files + their example counts (for sanity check).""" + out: Dict[str, int] = {} + if not os.path.isdir(data_dir): + return out + for fname in sorted(os.listdir(data_dir)): + if not fname.endswith((".jsonl", ".ndjson")): + continue + path = os.path.join(data_dir, fname) + with open(path, "r", encoding="utf-8") as f: + out[fname] = sum(1 for line in f if line.strip()) + return out diff --git a/nexus/training/trainer.py b/nexus/training/trainer.py new file mode 100644 index 0000000000000000000000000000000000000000..a41d141aa107ae559883c0594546cc04ad65629b --- /dev/null +++ b/nexus/training/trainer.py @@ -0,0 +1,266 @@ +""" +Nexus Trainer - Training loop cho Nexus Coder +============================================== +Hỗ trợ: +- Training với kiến trúc MoE +- Auxiliary loss (load balancing) +- Checkpointing +- Logging +- Mixed precision (fp16/bf16) +""" +import os +import json +import time +import torch +import torch.nn as nn +from torch.utils.data import DataLoader +from torch.optim import AdamW +from torch.optim.lr_scheduler import LambdaLR +from typing import Optional, Dict, Callable +from tqdm.auto import tqdm + +from ..model.nexus_coder import NexusCoderForCausalLM +from ..config import NexusConfig +from .dataset import NexusDataset, AUTHOR_TRAINING_DATA + + +def get_cosine_schedule_with_warmup( + optimizer, + num_warmup_steps: int, + num_training_steps: int, + num_cycles: float = 0.5, + last_epoch: int = -1, +): + """Cosine LR schedule với warmup.""" + def lr_lambda(current_step): + if current_step < num_warmup_steps: + return float(current_step) / float(max(1, num_warmup_steps)) + progress = float(current_step - num_warmup_steps) / float( + max(1, num_training_steps - num_warmup_steps) + ) + return max(0.0, 0.5 * (1.0 + __import__("math").cos(__import__("math").pi * num_cycles * 2.0 * progress))) + + return LambdaLR(optimizer, lr_lambda, last_epoch) + + +class NexusTrainer: + """Trainer cho Nexus Coder.""" + + def __init__( + self, + model: NexusCoderForCausalLM, + config: NexusConfig, + train_dataset: NexusDataset, + output_dir: str = "./checkpoints", + learning_rate: float = 5e-4, + weight_decay: float = 0.01, + warmup_steps: int = 100, + max_steps: int = 5000, + per_device_batch_size: int = 4, + gradient_accumulation_steps: int = 4, + logging_steps: int = 10, + save_steps: int = 500, + use_amp: bool = False, + amp_dtype: torch.dtype = torch.float16, + ): + self.model = model + self.config = config + self.train_dataset = train_dataset + self.output_dir = output_dir + self.learning_rate = learning_rate + self.weight_decay = weight_decay + self.warmup_steps = warmup_steps + self.max_steps = max_steps + self.per_device_batch_size = per_device_batch_size + self.gradient_accumulation_steps = gradient_accumulation_steps + self.logging_steps = logging_steps + self.save_steps = save_steps + self.use_amp = use_amp + self.amp_dtype = amp_dtype + + # Device + self.device = torch.device("cuda" if torch.cuda.is_available() else "cpu") + self.model = self.model.to(self.device) + + # DataLoader + self.dataloader = DataLoader( + train_dataset, + batch_size=per_device_batch_size, + shuffle=True, + num_workers=0, + pin_memory=True, + ) + + # Optimizer + self.optimizer = AdamW( + model.parameters(), + lr=learning_rate, + weight_decay=weight_decay, + betas=(0.9, 0.95), + eps=1e-8, + ) + + # Scheduler + total_steps = max_steps + self.scheduler = get_cosine_schedule_with_warmup( + self.optimizer, + num_warmup_steps=warmup_steps, + num_training_steps=total_steps, + ) + + # AMP scaler + self.scaler = torch.amp.GradScaler("cuda") if use_amp and torch.cuda.is_available() else None + + # Logging + os.makedirs(output_dir, exist_ok=True) + self.log_history = [] + + def train(self, resume_from_checkpoint: Optional[str] = None) -> Dict: + """Bắt đầu training.""" + print("=" * 60) + print(f" Nexus Coder Training") + print(f" Tác giả: {self.config.author}") + print(f" Device: {self.device}") + print(f" Steps: {self.max_steps}") + print("=" * 60) + + global_step = 0 + if resume_from_checkpoint and os.path.exists(resume_from_checkpoint): + global_step = self._load_checkpoint(resume_from_checkpoint) + + self.model.train() + start_time = time.time() + + # Training loop + dataloader_iter = iter(self.dataloader) + accumulated_loss = 0.0 + + progress_bar = tqdm(range(global_step, self.max_steps), desc="Training") + for step in progress_bar: + try: + batch = next(dataloader_iter) + except StopIteration: + dataloader_iter = iter(self.dataloader) + batch = next(dataloader_iter) + + input_ids = batch["input_ids"].to(self.device) + labels = batch["labels"].to(self.device) + attention_mask = batch["attention_mask"].to(self.device) + + # Forward + if self.use_amp and torch.cuda.is_available(): + with torch.amp.autocast("cuda", dtype=self.amp_dtype): + outputs = self.model( + input_ids=input_ids, + attention_mask=attention_mask, + labels=labels, + ) + loss = outputs["loss"] / self.gradient_accumulation_steps + self.scaler.scale(loss).backward() + else: + outputs = self.model( + input_ids=input_ids, + attention_mask=attention_mask, + labels=labels, + ) + loss = outputs["loss"] / self.gradient_accumulation_steps + loss.backward() + + accumulated_loss += loss.item() + + # Optimizer step + if (step + 1) % self.gradient_accumulation_steps == 0: + if self.use_amp and torch.cuda.is_available(): + self.scaler.unscale_(self.optimizer) + torch.nn.utils.clip_grad_norm_(self.model.parameters(), 1.0) + self.scaler.step(self.optimizer) + self.scaler.update() + else: + torch.nn.utils.clip_grad_norm_(self.model.parameters(), 1.0) + self.optimizer.step() + self.optimizer.zero_grad() + self.scheduler.step() + global_step += 1 + + # Logging + if global_step % self.logging_steps == 0: + avg_loss = accumulated_loss / self.logging_steps + elapsed = time.time() - start_time + lr = self.scheduler.get_last_lr()[0] + log_entry = { + "step": global_step, + "loss": avg_loss, + "learning_rate": lr, + "elapsed_seconds": elapsed, + } + self.log_history.append(log_entry) + progress_bar.set_postfix({ + "loss": f"{avg_loss:.4f}", + "lr": f"{lr:.2e}", + }) + accumulated_loss = 0.0 + + # Save checkpoint + if global_step % self.save_steps == 0: + self._save_checkpoint(global_step) + + if global_step >= self.max_steps: + break + + # Final save + self._save_checkpoint(global_step, final=True) + + # Save log + self._save_logs() + + elapsed = time.time() - start_time + print(f"\n✅ Training hoàn thành trong {elapsed:.1f}s") + return {"global_step": global_step, "elapsed": elapsed} + + def _save_checkpoint(self, step: int, final: bool = False) -> None: + """Lưu checkpoint (v0.4: include AMP scaler state for safe resume).""" + suffix = "final" if final else f"step-{step}" + path = os.path.join(self.output_dir, f"nexus_coder-{suffix}.pt") + # v0.4 fix: persist GradScaler state so AMP can resume safely without + # scale-factor NaNs on first few steps. + scaler_state = None + scaler = getattr(self, "scaler", None) + if scaler is not None and hasattr(scaler, "state_dict"): + try: + scaler_state = scaler.state_dict() + except Exception: + scaler_state = None + torch.save({ + "model_state_dict": self.model.state_dict(), + "optimizer_state_dict": self.optimizer.state_dict(), + "scheduler_state_dict": self.scheduler.state_dict(), + "scaler_state_dict": scaler_state, + "step": step, + "config": self.config.__dict__, + }, path) + print(f" Checkpoint saved: {path}") + + def _load_checkpoint(self, path: str) -> int: + """Load checkpoint (v0.4: also restore AMP scaler if present).""" + checkpoint = torch.load(path, map_location=self.device, weights_only=False) + self.model.load_state_dict(checkpoint["model_state_dict"]) + self.optimizer.load_state_dict(checkpoint["optimizer_state_dict"]) + self.scheduler.load_state_dict(checkpoint["scheduler_state_dict"]) + # v0.4 fix: restore scaler state if present + scaler_state = checkpoint.get("scaler_state_dict") + scaler = getattr(self, "scaler", None) + if scaler_state is not None and scaler is not None and hasattr(scaler, "load_state_dict"): + try: + scaler.load_state_dict(scaler_state) + except Exception: + pass + step = checkpoint.get("step", 0) + print(f" Resumed from checkpoint at step {step}") + return step + + def _save_logs(self) -> None: + """Lưu training logs.""" + log_path = os.path.join(self.output_dir, "training_log.json") + with open(log_path, "w", encoding="utf-8") as f: + json.dump(self.log_history, f, ensure_ascii=False, indent=2) + print(f" 📝 Saved training log: {log_path}") diff --git a/nexus/utils/__init__.py b/nexus/utils/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..c92c7ecae98b5caeaa7a3356eb682af2bd982d20 --- /dev/null +++ b/nexus/utils/__init__.py @@ -0,0 +1,4 @@ +"""Utils package cho Nexus Coder.""" +from .logging import get_logger + +__all__ = ["get_logger"] diff --git a/nexus/utils/logging.py b/nexus/utils/logging.py new file mode 100644 index 0000000000000000000000000000000000000000..2b70fe86497ebee6b1ffafead8d51ac84eeca27d --- /dev/null +++ b/nexus/utils/logging.py @@ -0,0 +1,22 @@ +"""Logging utilities cho Nexus Coder.""" +import logging +import sys +from typing import Optional + + +def get_logger(name: str = "nexus", level: int = logging.INFO) -> logging.Logger: + """Tạo logger chuẩn cho Nexus.""" + logger = logging.getLogger(name) + if logger.handlers: + return logger + + logger.setLevel(level) + handler = logging.StreamHandler(sys.stdout) + handler.setFormatter( + logging.Formatter( + "%(asctime)s [%(name)s] %(levelname)s: %(message)s", + datefmt="%Y-%m-%d %H:%M:%S", + ) + ) + logger.addHandler(handler) + return logger diff --git a/pyproject.toml b/pyproject.toml new file mode 100644 index 0000000000000000000000000000000000000000..1bd72f47365a70c63dd3983f588ed905a6fa9b19 --- /dev/null +++ b/pyproject.toml @@ -0,0 +1,108 @@ +[build-system] +requires = ["setuptools>=68.0", "wheel"] +build-backend = "setuptools.build_meta" + +[project] +name = "nexus-coder" +version = "0.4.0" +description = "Nexus Coder v0.4 - CyberForge edition. MoE 423B/39B + 3M context + CyberGym training. Created by Hieu Louis." +readme = "README.md" +license = {file = "LICENSE"} +authors = [ + {name = "Hieu Louis", email = "mhieuhonda@users.noreply.github.com"} +] +requires-python = "==3.12.13" +dependencies = [ + "torch>=2.0.0", + "numpy>=1.24.0", + "tqdm>=4.65.0", + "pyyaml>=6.0", + "datasets>=2.14.0", + "requests>=2.31.0", + "cryptography>=41.0.0", +] +keywords = ["ai", "llm", "moe", "mixture-of-experts", "transformer", "nexus", "agent", "skills", "tools", "cyberforge", "cybergym"] +classifiers = [ + "Development Status :: 4 - Beta", + "Intended Audience :: Developers", + "Intended Audience :: Science/Research", + "License :: Other/Proprietary License", + "Programming Language :: Python :: 3.12", + "Topic :: Scientific/Engineering :: Artificial Intelligence", +] + +[project.optional-dependencies] +dev = ["pytest>=7.0", "black", "flake8", "ruff"] +gpu = ["flash-attn>=2.0.0", "bitsandbytes>=0.41.0", "triton>=2.0.0"] +data = ["datasets>=2.14.0", "datasketch>=1.6.0", "langdetect>=1.0.9"] +tools = ["ruff>=0.1.0", "black>=23.0.0", "isort>=5.12.0", "sqlparse>=0.4.4", "jsbeautifier>=1.14.0"] +crypto = ["cryptography>=41.0.0", "pyjwt>=2.8.0"] +database = [ + "sqlalchemy>=2.0.0", + "psycopg2-binary>=2.9.0", + "pymysql>=1.1.0", + "redis>=5.0.0", + "pymongo>=4.5.0", + "elasticsearch>=8.0.0", + "kafka-python>=2.0.2", + "pika>=1.3.0", +] +web = [ + "aiohttp>=3.9.0", + "websockets>=12.0", + "grpcio>=1.59.0", + "beautifulsoup4>=4.12.0", + "lxml>=4.9.0", +] +devops = ["paramiko>=3.4.0", "kubernetes>=28.1.0", "docker>=7.0.0"] +media = ["Pillow>=10.0.0", "reportlab>=4.0.0", "markdown>=3.5.0"] +ml = ["scikit-learn>=1.3.0", "scipy>=1.11.0", "transformers>=4.35.0", "accelerate>=0.24.0", "peft>=0.6.0", "lm-eval>=0.3.0"] +distributed = ["deepspeed>=0.12.0", "accelerate>=0.24.0", "flash-attn>=2.0.0"] +all = [ + "flash-attn>=2.0.0", + "bitsandbytes>=0.41.0", + "triton>=2.0.0", + "datasets>=2.14.0", + "datasketch>=1.6.0", + "langdetect>=1.0.9", + "ruff>=0.1.0", + "black>=23.0.0", + "isort>=5.12.0", + "sqlparse>=0.4.4", + "jsbeautifier>=1.14.0", + "cryptography>=41.0.0", + "pyjwt>=2.8.0", + "aiohttp>=3.9.0", + "websockets>=12.0", + "grpcio>=1.59.0", + "beautifulsoup4>=4.12.0", + "lxml>=4.9.0", + "sqlalchemy>=2.0.0", + "psycopg2-binary>=2.9.0", + "pymysql>=1.1.0", + "redis>=5.0.0", + "pymongo>=4.5.0", + "elasticsearch>=8.0.0", + "kafka-python>=2.0.2", + "pika>=1.3.0", + "paramiko>=3.4.0", + "kubernetes>=28.1.0", + "docker>=7.0.0", + "Pillow>=10.0.0", + "reportlab>=4.0.0", + "markdown>=3.5.0", + "scikit-learn>=1.3.0", + "scipy>=1.11.0", + "transformers>=4.35.0", + "accelerate>=0.24.0", + "peft>=0.6.0", +] + +[project.urls] +Homepage = "https://github.com/mhieuhonda/NexusCoder" +Repository = "https://github.com/mhieuhonda/NexusCoder" +Issues = "https://github.com/mhieuhonda/NexusCoder/issues" +Changelog = "https://github.com/mhieuhonda/NexusCoder/blob/main/CHANGELOG.md" + +[tool.setuptools.packages.find] +include = ["nexus*"] diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000000000000000000000000000000000000..95d787fc0f081e4c01529bb971b72d2119ddf734 --- /dev/null +++ b/requirements.txt @@ -0,0 +1,75 @@ +# Requirements for Nexus Coder v0.3 +# Python 3.12.13 (strict) +# Author: Hieu Louis (2026) + +# === Core deep learning === +torch>=2.0.0 +numpy>=1.24.0 + +# === Utilities === +tqdm>=4.65.0 +pyyaml>=6.0 + +# === Data pipeline (v0.2 + v0.3 NEW) === +datasets>=2.14.0 +datasketch>=1.6.0 +langdetect>=1.0.9 + +# === Code tools (v0.2) === +ruff>=0.1.0 +black>=23.0.0 +isort>=5.12.0 +flake8>=6.0.0 +sqlparse>=0.4.4 +jsbeautifier>=1.14.0 + +# === Optimization (v0.2) === +bitsandbytes>=0.41.0; platform_system == "Linux" + +# === Web tools (v0.2 + v0.3 NEW) === +requests>=2.31.0 +aiohttp>=3.9.0 +websockets>=12.0 +grpcio>=1.59.0 +beautifulsoup4>=4.12.0 +lxml>=4.9.0 + +# === Crypto / Security tools (v0.2 + v0.3 NEW) === +cryptography>=41.0.0 +pyjwt>=2.8.0 + +# === Database tools (v0.3 NEW) === +sqlalchemy>=2.0.0 +psycopg2-binary>=2.9.0 +pymysql>=1.1.0 +redis>=5.0.0 +pymongo>=4.5.0 +elasticsearch>=8.0.0 +kafka-python>=2.0.2 +pika>=1.3.0 + +# === DevOps tools (v0.3 NEW) === +paramiko>=3.4.0 +kubernetes>=28.1.0 +docker>=7.0.0 + +# === Media / Convert tools (v0.3 NEW) === +Pillow>=10.0.0 +reportlab>=4.0.0 +markdown>=3.5.0 + +# === ML tools (v0.3 NEW) === +scikit-learn>=1.3.0 +scipy>=1.11.0 +transformers>=4.35.0 +accelerate>=0.24.0 +peft>=0.6.0 + +# === Optional advanced features === +# For distributed training +# deepspeed>=0.12.0 +# For faster attention (GPU only) +# flash-attn>=2.0.0 +# triton>=2.0.0 +# For evaluation benchmarks +# lm-eval>=0.3.0 diff --git a/scripts/chat.py b/scripts/chat.py new file mode 100644 index 0000000000000000000000000000000000000000..5fdc3f3c3a53d611d76aa5bf90e0e7b3bc78a943 --- /dev/null +++ b/scripts/chat.py @@ -0,0 +1,83 @@ +""" +Script chat với Nexus Agent +============================= +Chạy: python scripts/chat.py +""" +import sys +import os +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +import torch + +from nexus.config import NexusConfig +from nexus.model.nexus_coder import NexusCoderForCausalLM +from nexus.tokenizer.tokenizer import NexusTokenizer +from nexus.inference.generator import NexusGenerator +from nexus.agent.agent import NexusAgent +from nexus.training.dataset import AUTHOR_TRAINING_DATA + + +def get_tiny_config() -> NexusConfig: + """Tiny config cho demo chat.""" + return NexusConfig( + vocab_size=2000, + hidden_size=256, + num_hidden_layers=4, + num_attention_heads=8, + num_kv_heads=2, + head_dim=32, + intermediate_size=512, + num_experts=4, + num_active_experts=2, + max_position_embeddings=512, + ) + + +def main(): + print("=" * 60) + print(" NEXUS CODER v0.1 - Chat Demo") + print(" Tác giả: Hieu Louis") + print(" Năm: 2026") + print("=" * 60) + + # Init config (dùng tiny cho demo, vì full 10B cần GPU) + config = get_tiny_config() + print(f"\n📝 Cấu hình demo: hidden={config.hidden_size}, layers={config.num_hidden_layers}") + + # Tokenizer + print("\n🔨 Đang huấn luyện tokenizer...") + tokenizer = NexusTokenizer(vocab_size=config.vocab_size) + corpus = [f"{d['system']} {d['user']} {d['assistant']}" for d in AUTHOR_TRAINING_DATA] + tokenizer.train(corpus) + print(f" ✓ {tokenizer.vocab_size} tokens") + + # Model + print("\n🧠 Đang khởi tạo model...") + model = NexusCoderForCausalLM(config) + print(" ✓ Model ready (random weights - đây chỉ là demo kiến trúc)") + + # Generator + generator = NexusGenerator( + model=model, + tokenizer=tokenizer, + config=config, + ) + + # Agent + agent = NexusAgent( + generator=generator, + config=config, + name="Nexus", + personality="humorous", + language="bilingual", + ) + + # Print info + agent._print_info() + + # Start chat + agent.chat() + + +if __name__ == "__main__": + main() diff --git a/scripts/collect_data.py b/scripts/collect_data.py new file mode 100644 index 0000000000000000000000000000000000000000..01df436214f5e406336bdf862262450eeb315fbb --- /dev/null +++ b/scripts/collect_data.py @@ -0,0 +1,210 @@ +""" +Script thu thập training data từ GitHub + HuggingFace +===================================================== +Chạy script này để collect training data cho Nexus Coder v0.2. + +Sources: +- GitHub repos (curated list trong nexus.data.collectors.github_collector.CURATED_REPOS) +- HuggingFace datasets (curated list trong nexus.data.collectors.huggingface_collector.CURATED_DATASETS) +- arXiv papers (curated queries) +- Wikipedia (Vietnamese + English) +- StackOverflow Q&A + +Usage: + python scripts/collect_data.py --source github --max-repos 10 + python scripts/collect_data.py --source huggingface --max-datasets 5 + python scripts/collect_data.py --source all --output ./data/raw +""" +import sys +import os +import argparse +import json +import logging +from pathlib import Path + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +logging.basicConfig( + level=logging.INFO, + format="%(asctime)s [%(levelname)s] %(message)s", +) +logger = logging.getLogger(__name__) + + +def collect_github(output_dir: str, max_repos: int = 10, token: str = None): + """Collect code từ GitHub repos.""" + from nexus.data.collectors.github_collector import GitHubCollector, CURATED_REPOS + + collector = GitHubCollector(token=token, cache_dir=os.path.join(output_dir, "github_cache")) + repos = CURATED_REPOS[:max_repos] + + logger.info(f"Collecting from {len(repos)} GitHub repos...") + + output_file = os.path.join(output_dir, "github_code.jsonl") + count = 0 + + with open(output_file, "w", encoding="utf-8") as f: + for sample in collector.collect(repos): + entry = { + "text": sample.content, + "source": f"github:{sample.repo}", + "language": sample.language, + "metadata": { + "file_path": sample.file_path, + "size": sample.size, + "quality_score": sample.quality_score, + }, + } + f.write(json.dumps(entry, ensure_ascii=False) + "\n") + count += 1 + + if count % 100 == 0: + logger.info(f" Collected {count} samples...") + + logger.info(f"✓ GitHub: {count} samples → {output_file}") + return count + + +def collect_huggingface(output_dir: str, max_datasets: int = 5, token: str = None): + """Collect từ HuggingFace datasets.""" + from nexus.data.collectors.huggingface_collector import HuggingFaceCollector, CURATED_DATASETS + + collector = HuggingFaceCollector(cache_dir=os.path.join(output_dir, "hf_cache"), token=token) + datasets = CURATED_DATASETS[:max_datasets] + + logger.info(f"Collecting from {len(datasets)} HuggingFace datasets...") + + output_file = os.path.join(output_dir, "hf_data.jsonl") + count = 0 + + with open(output_file, "w", encoding="utf-8") as f: + for sample in collector.collect(datasets): + f.write(json.dumps(sample, ensure_ascii=False) + "\n") + count += 1 + + if count % 1000 == 0: + logger.info(f" Collected {count} samples...") + + logger.info(f"✓ HuggingFace: {count} samples → {output_file}") + return count + + +def collect_arxiv(output_dir: str, max_queries: int = 5): + """Collect papers từ arXiv.""" + from nexus.data.collectors.arxiv_collector import ArxivCollector, CURATED_QUERIES + + collector = ArxivCollector() + queries = CURATED_QUERIES[:max_queries] + + logger.info(f"Collecting arXiv papers ({len(queries)} queries)...") + + output_file = os.path.join(output_dir, "arxiv_papers.jsonl") + count = 0 + + with open(output_file, "w", encoding="utf-8") as f: + for sample in collector.collect(queries, max_per_query=20): + f.write(json.dumps(sample, ensure_ascii=False) + "\n") + count += 1 + + logger.info(f"✓ arXiv: {count} samples → {output_file}") + return count + + +def collect_wikipedia(output_dir: str, language: str = "vi"): + """Collect articles từ Wikipedia.""" + from nexus.data.collectors.wikipedia_collector import WikipediaCollector + + collector = WikipediaCollector(language=language) + + logger.info(f"Collecting Wikipedia ({language}) articles...") + + output_file = os.path.join(output_dir, f"wikipedia_{language}.jsonl") + count = 0 + + with open(output_file, "w", encoding="utf-8") as f: + for sample in collector.collect(): + f.write(json.dumps(sample, ensure_ascii=False) + "\n") + count += 1 + + logger.info(f"✓ Wikipedia ({language}): {count} samples → {output_file}") + return count + + +def collect_stackoverflow(output_dir: str, max_tags: int = 5, token: str = None): + """Collect Q&A từ StackOverflow.""" + from nexus.data.collectors.stackoverflow_collector import StackOverflowCollector, CURATED_TAGS + + collector = StackOverflowCollector(key=token) + tags = CURATED_TAGS[:max_tags] + + logger.info(f"Collecting StackOverflow Q&A ({len(tags)} tags)...") + + output_file = os.path.join(output_dir, "stackoverflow.jsonl") + count = 0 + + with open(output_file, "w", encoding="utf-8") as f: + for sample in collector.collect(tags, max_per_tag=50): + f.write(json.dumps(sample, ensure_ascii=False) + "\n") + count += 1 + + logger.info(f"✓ StackOverflow: {count} samples → {output_file}") + return count + + +def main(): + parser = argparse.ArgumentParser(description="Nexus Coder Data Collector") + parser.add_argument( + "--source", + choices=["github", "huggingface", "arxiv", "wikipedia", "stackoverflow", "all"], + default="all", + help="Data source to collect from", + ) + parser.add_argument( + "--output", + type=str, + default="./data/raw", + help="Output directory", + ) + parser.add_argument("--max-repos", type=int, default=10, help="Max GitHub repos") + parser.add_argument("--max-datasets", type=int, default=5, help="Max HF datasets") + parser.add_argument("--max-queries", type=int, default=5, help="Max arXiv queries") + parser.add_argument("--max-tags", type=int, default=5, help="Max SO tags") + parser.add_argument("--language", type=str, default="vi", help="Wikipedia language") + parser.add_argument("--github-token", type=str, default=os.environ.get("GITHUB_TOKEN")) + parser.add_argument("--hf-token", type=str, default=os.environ.get("HF_TOKEN")) + + args = parser.parse_args() + + print("=" * 70) + print(" NEXUS CODER v0.2 - DATA COLLECTOR") + print(" Tác giả: Hieu Louis") + print("=" * 70) + + os.makedirs(args.output, exist_ok=True) + + total = 0 + + if args.source in ("github", "all"): + total += collect_github(args.output, args.max_repos, args.github_token) + + if args.source in ("huggingface", "all"): + total += collect_huggingface(args.output, args.max_datasets, args.hf_token) + + if args.source in ("arxiv", "all"): + total += collect_arxiv(args.output, args.max_queries) + + if args.source in ("wikipedia", "all"): + total += collect_wikipedia(args.output, args.language) + + if args.source in ("stackoverflow", "all"): + total += collect_stackoverflow(args.output, args.max_tags) + + print(f"\n{'=' * 70}") + print(f" ✅ Total collected: {total} samples") + print(f" 📁 Output: {args.output}") + print(f"{'=' * 70}") + print(f"\nNext step: Run scripts/prepare_dataset.py to process the raw data.") + + +if __name__ == "__main__": + main() diff --git a/scripts/count_params.py b/scripts/count_params.py new file mode 100644 index 0000000000000000000000000000000000000000..d66571f5c7b8886ee998a172984041d79db54224 --- /dev/null +++ b/scripts/count_params.py @@ -0,0 +1,42 @@ +""" +Script đếm tham số Nexus Coder 10B / 1.5B active +================================================= +Chạy: python scripts/count_params.py +""" +import sys +import os +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +from nexus.config import NexusConfig, print_config_summary + + +def main(): + """In tóm tắt cấu hình và tham số.""" + config = NexusConfig() + print_config_summary(config) + + stats = config.estimated_total_params() + + print("\n📋 Chi tiết tính toán tham số:") + print(f" Embedding (vocab×hidden): {stats['embedding']:,} ({stats['embedding']/1e6:.1f}M)") + print(f" Attention per layer: {stats['attention_per_layer']:,} ({stats['attention_per_layer']/1e6:.1f}M)") + print(f" MoE per layer (total): {stats['moe_total_per_layer']:,} ({stats['moe_total_per_layer']/1e6:.1f}M)") + print(f" MoE per layer (active): {stats['moe_active_per_layer']:,} ({stats['moe_active_per_layer']/1e6:.1f}M)") + print(f" Router per layer: {stats['router_per_layer']:,}") + print(f" Per layer (total): {stats['per_layer_total']:,} ({stats['per_layer_total']/1e6:.1f}M)") + print(f" Per layer (active): {stats['per_layer_active']:,} ({stats['per_layer_active']/1e6:.1f}M)") + print(f" Số layers: {stats['total_layers']}") + print() + print(f" ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━") + print(f" Tổng tham số: {stats['total_params']:>15,} ({stats['total_params_billion']:.2f}B)") + print(f" Tham số active:{stats['active_params']:>15,} ({stats['active_params_billion']:.2f}B)") + print(f" ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━") + + # Verify + assert 9.5e9 < stats["total_params"] < 11e9, "❌ Total params không đúng (phải ~10B)" + assert 1.3e9 < stats["active_params"] < 1.7e9, "❌ Active params không đúng (phải ~1.5B)" + print("\n✅ Đã xác nhận: 10B tổng tham số / 1.5B tham số active - đúng theo yêu cầu!") + + +if __name__ == "__main__": + main() diff --git a/scripts/evaluate.py b/scripts/evaluate.py new file mode 100644 index 0000000000000000000000000000000000000000..c8b170620e88e4b0eef938bded97f10bc208e274 --- /dev/null +++ b/scripts/evaluate.py @@ -0,0 +1,81 @@ +""" +Script đánh giá model trên benchmarks +====================================== +Usage: + python scripts/evaluate.py --model model.pt --benchmarks humaneval,gsm8k + python scripts/evaluate.py --model model.pt --benchmarks all --sample-size 100 +""" +import sys +import os +import argparse +import json + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +import torch + +from nexus.config import get_config_by_name +from nexus.model.nexus_coder import NexusCoderForCausalLM +from nexus.tokenizer.tokenizer import NexusTokenizer +from nexus.eval.benchmarks import BenchmarkSuite +from nexus.eval.metrics import compute_perplexity + + +def main(): + parser = argparse.ArgumentParser(description="Nexus Coder Evaluator") + parser.add_argument("--model", type=str, required=True, help="Path to model checkpoint") + parser.add_argument("--config", type=str, default="large", help="Model config") + parser.add_argument( + "--benchmarks", + type=str, + default="humaneval", + help="Comma-separated benchmark names", + ) + parser.add_argument("--sample-size", type=int, default=None, help="Limit examples per benchmark") + parser.add_argument("--output", type=str, default="./eval_results.json", help="Output file") + + args = parser.parse_args() + + print("=" * 60) + print(" NEXUS CODER v0.2 - EVALUATION") + print("=" * 60) + + # Load model + config = get_config_by_name(args.config) + model = NexusCoderForCausalLM(config) + + if os.path.exists(args.model): + checkpoint = torch.load(args.model, map_location="cpu", weights_only=False) + if "model_state_dict" in checkpoint: + model.load_state_dict(checkpoint["model_state_dict"]) + else: + model.load_state_dict(checkpoint) + print(f"✓ Loaded model from {args.model}") + else: + print(f"⚠️ Model file not found, using random init: {args.model}") + + # Tokenizer + tokenizer = NexusTokenizer() + + # Benchmarks + benchmarks = args.benchmarks.split(",") if args.benchmarks != "all" else None + + suite = BenchmarkSuite() + print(f"\n📋 Available benchmarks: {len(suite.list_available())}") + for b in suite.list_available(): + print(f" - {b.name}: {b.description}") + + print(f"\n🏃 Running benchmarks: {benchmarks or 'all'}") + results = suite.run(model, tokenizer, benchmarks=benchmarks, sample_size=args.sample_size) + + # Save results + with open(args.output, "w", encoding="utf-8") as f: + json.dump(results, f, indent=2, ensure_ascii=False, default=str) + + print(f"\n📊 Results:") + print(suite.summary()) + print(f"\n💾 Saved to: {args.output}") + + +if __name__ == "__main__": + main() diff --git a/scripts/prepare_dataset.py b/scripts/prepare_dataset.py new file mode 100644 index 0000000000000000000000000000000000000000..d0045f7d88658f31488ecf048e57763da9512f6c --- /dev/null +++ b/scripts/prepare_dataset.py @@ -0,0 +1,176 @@ +""" +Script chuẩn bị dataset cho training +==================================== +Process raw collected data → cleaned, deduplicated, formatted training data. + +Steps: +1. Load raw data from ./data/raw/ +2. Clean text (TextCleaner) +3. Format code samples (CodeFormatter) +4. Filter by quality (QualityFilter) +5. Deduplicate (Deduplicator) +6. Save processed data to ./data/processed/ + +Usage: + python scripts/prepare_dataset.py --input ./data/raw --output ./data/processed +""" +import sys +import os +import json +import argparse +import logging +from pathlib import Path + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +logging.basicConfig(level=logging.INFO, format="%(asctime)s [%(levelname)s] %(message)s") +logger = logging.getLogger(__name__) + + +def load_raw_data(input_dir: str): + """Load all JSONL files from input directory.""" + files = [ + f for f in os.listdir(input_dir) + if f.endswith(".jsonl") + ] + + total = 0 + for fname in files: + fpath = os.path.join(input_dir, fname) + count = 0 + with open(fpath, "r", encoding="utf-8") as f: + for line in f: + try: + item = json.loads(line) + yield item + count += 1 + except json.JSONDecodeError: + continue + logger.info(f" Loaded {count} from {fname}") + total += count + + logger.info(f"Total raw samples: {total}") + + +def process_data(input_dir: str, output_dir: str, max_samples: int = None): + """Process raw data through cleaning, dedup, quality filter.""" + from nexus.data.processors.cleaner import TextCleaner + from nexus.data.processors.quality_filter import QualityFilter + from nexus.data.processors.code_formatter import CodeFormatter + from nexus.data.processors.deduplicator import Deduplicator + from nexus.data.curriculum import CurriculumLearning + + cleaner = TextCleaner() + quality_filter = QualityFilter() + code_formatter = CodeFormatter() + deduplicator = Deduplicator() + curriculum = CurriculumLearning() + + os.makedirs(output_dir, exist_ok=True) + + # Output files by difficulty + output_files = { + "easy": open(os.path.join(output_dir, "train_easy.jsonl"), "w", encoding="utf-8"), + "medium": open(os.path.join(output_dir, "train_medium.jsonl"), "w", encoding="utf-8"), + "hard": open(os.path.join(output_dir, "train_hard.jsonl"), "w", encoding="utf-8"), + "expert": open(os.path.join(output_dir, "train_expert.jsonl"), "w", encoding="utf-8"), + } + + stats = { + "total_input": 0, + "cleaned": 0, + "quality_passed": 0, + "deduplicated": 0, + "by_difficulty": {"easy": 0, "medium": 0, "hard": 0, "expert": 0}, + } + + logger.info("Processing samples...") + + for sample in load_raw_data(input_dir): + if max_samples and stats["total_input"] >= max_samples: + break + + stats["total_input"] += 1 + + # Step 1: Clean + sample = cleaner.process(sample) + if sample is None: + continue + stats["cleaned"] += 1 + + # Step 2: Format code + sample = code_formatter.process(sample) + + # Step 3: Quality filter + if not quality_filter.filter(sample): + continue + sample = next(quality_filter.process([sample]), None) + if sample is None: + continue + stats["quality_passed"] += 1 + + # Step 4: Dedup + if deduplicator.is_duplicate(sample.get("text", "")): + continue + deduplicator.add(sample["text"], sample) + stats["deduplicated"] += 1 + + # Step 5: Classify by difficulty + difficulty = curriculum.classify_sample(sample).value + output_files[difficulty].write(json.dumps(sample, ensure_ascii=False) + "\n") + stats["by_difficulty"][difficulty] += 1 + + if stats["deduplicated"] % 1000 == 0: + logger.info(f" Processed {stats['deduplicated']} unique samples...") + + # Close files + for f in output_files.values(): + f.close() + + # Print stats + print("\n" + "=" * 60) + print(" PROCESSING COMPLETE") + print("=" * 60) + print(f" Input samples: {stats['total_input']:,}") + print(f" After cleaning: {stats['cleaned']:,}") + print(f" Quality passed: {stats['quality_passed']:,}") + print(f" After dedup: {stats['deduplicated']:,}") + print("-" * 60) + print(" By difficulty:") + for level, count in stats["by_difficulty"].items(): + print(f" {level:8s}: {count:,}") + print("-" * 60) + print(f" Output dir: {output_dir}") + print("=" * 60) + + # Save stats + stats_path = os.path.join(output_dir, "processing_stats.json") + with open(stats_path, "w", encoding="utf-8") as f: + json.dump(stats, f, indent=2) + + return stats + + +def main(): + parser = argparse.ArgumentParser(description="Nexus Coder Dataset Processor") + parser.add_argument("--input", type=str, default="./data/raw") + parser.add_argument("--output", type=str, default="./data/processed") + parser.add_argument("--max-samples", type=int, default=None) + args = parser.parse_args() + + print("=" * 70) + print(" NEXUS CODER v0.2 - DATASET PROCESSOR") + print(" Tác giả: Hieu Louis") + print("=" * 70) + + if not os.path.exists(args.input): + print(f"\n❌ Input dir not found: {args.input}") + print("Run scripts/collect_data.py first to collect raw data.") + return 1 + + process_data(args.input, args.output, args.max_samples) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/quantize_model.py b/scripts/quantize_model.py new file mode 100644 index 0000000000000000000000000000000000000000..7cb0638a6d69c2e62e38ccd10a209e58f4e241f2 --- /dev/null +++ b/scripts/quantize_model.py @@ -0,0 +1,86 @@ +""" +Script quantize model cho inference +==================================== +Quantize Nexus Coder model để giảm memory footprint. + +Usage: + python scripts/quantize_model.py --input model.pt --method int8 --output model_int8.pt + python scripts/quantize_model.py --input model.pt --method int4 --output model_int4.pt +""" +import sys +import os +import argparse + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +import torch + +from nexus.config import NexusConfig +from nexus.model.nexus_coder import NexusCoderForCausalLM +from nexus.optim.quantization import Quantizer, QuantizationConfig + + +def main(): + parser = argparse.ArgumentParser(description="Nexus Coder Quantizer") + parser.add_argument("--input", type=str, required=True, help="Path to model checkpoint") + parser.add_argument("--output", type=str, required=True, help="Output path") + parser.add_argument( + "--method", + choices=["int8", "int4", "fp8"], + default="int8", + help="Quantization method", + ) + parser.add_argument("--config", type=str, default="large", help="Model config: tiny/small/medium/large/xlarge") + + args = parser.parse_args() + + print("=" * 60) + print(" NEXUS CODER v0.2 - MODEL QUANTIZER") + print("=" * 60) + + # Load config + from nexus.config import get_config_by_name + config = get_config_by_name(args.config) + + # Load model + print(f"\n📥 Loading model from {args.input}...") + model = NexusCoderForCausalLM(config) + + checkpoint = torch.load(args.input, map_location="cpu", weights_only=False) + if "model_state_dict" in checkpoint: + model.load_state_dict(checkpoint["model_state_dict"]) + else: + model.load_state_dict(checkpoint) + + # Estimate memory before + param_count = sum(p.numel() for p in model.parameters()) + fp16_mb = (param_count * 2) / (1024 * 1024) + print(f" Model: {param_count:,} params") + print(f" FP16 size: {fp16_mb:.0f} MB") + + # Quantize + print(f"\n🔧 Quantizing to {args.method.upper()}...") + quantizer = Quantizer(QuantizationConfig(method=args.method)) + quantized_model = quantizer.quantize(model) + + # Estimate memory after + estimates = quantizer.estimate_memory_savings(model) + print(f"\n📊 Memory estimates:") + print(f" FP16: {estimates['fp16_mb']:.0f} MB") + print(f" INT8: {estimates['int8_mb']:.0f} MB (savings: {estimates['int8_savings_pct']:.0f}%)") + print(f" INT4: {estimates['int4_mb']:.0f} MB (savings: {estimates['int4_savings_pct']:.0f}%)") + + # Save + print(f"\n💾 Saving quantized model to {args.output}...") + torch.save({ + "model_state_dict": quantized_model.state_dict(), + "config": config.__dict__, + "quantization": args.method, + }, args.output) + + output_size = os.path.getsize(args.output) / (1024 * 1024) + print(f"\n✅ Done! Output size: {output_size:.0f} MB") + + +if __name__ == "__main__": + main() diff --git a/scripts/quick_test.py b/scripts/quick_test.py new file mode 100644 index 0000000000000000000000000000000000000000..3665873226008a98c8824fc961f544d593069129 --- /dev/null +++ b/scripts/quick_test.py @@ -0,0 +1,181 @@ +""" +Verify architecture + counting tham số - chạy nhanh +""" +import sys +import os +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +import torch + +from nexus.config import NexusConfig, print_config_summary +from nexus.model.nexus_coder import NexusCoderForCausalLM + + +def test_tiny_model(): + """Test với model nhỏ.""" + print("\n[Test 1] Tiny model forward pass...") + tiny_config = NexusConfig( + vocab_size=1000, + hidden_size=128, + num_hidden_layers=2, + num_attention_heads=4, + num_kv_heads=2, + head_dim=32, + intermediate_size=256, + num_experts=4, + num_active_experts=2, + max_position_embeddings=512, + ) + model = NexusCoderForCausalLM(tiny_config) + + input_ids = torch.randint(0, 1000, (2, 16)) + labels = input_ids.clone() + + outputs = model(input_ids=input_ids, labels=labels) + assert outputs["loss"] is not None + assert outputs["logits"].shape == (2, 16, 1000) + print(f" ✓ Loss: {outputs['loss'].item():.4f}") + print(f" ✓ Logits shape: {outputs['logits'].shape}") + + # Generate + generated = model.generate( + input_ids=torch.randint(0, 1000, (1, 4)), + max_new_tokens=10, + do_sample=False, + ) + assert generated.shape[1] > 4 + print(f" ✓ Generated shape: {generated.shape}") + print(" ✓ PASSED!") + + +def test_param_count(): + """Test đếm tham số theo config.""" + print("\n[Test 2] Param count theo config...") + config = NexusConfig() + stats = config.estimated_total_params() + print(f" Total: {stats['total_params']:,} ({stats['total_params_billion']:.2f}B)") + print(f" Active: {stats['active_params']:,} ({stats['active_params_billion']:.2f}B)") + assert 9.5e9 < stats["total_params"] < 11e9 + assert 1.3e9 < stats["active_params"] < 1.7e9 + print(" ✓ PASSED!") + + +def test_tokenizer(): + """Test tokenizer cơ bản.""" + print("\n[Test 3] Tokenizer...") + from nexus.tokenizer.tokenizer import NexusTokenizer + from nexus.training.dataset import AUTHOR_TRAINING_DATA + + tokenizer = NexusTokenizer(vocab_size=2000) + corpus = [f"{d['system']} {d['user']} {d['assistant']}" for d in AUTHOR_TRAINING_DATA] + tokenizer.train(corpus) + + text = "Xin chào, tôi là Nexus Coder do Hieu Louis tạo ra." + ids = tokenizer.encode(text, add_special=True) + decoded = tokenizer.decode(ids) + + assert len(ids) > 0 + assert "Nexus" in decoded or "nexus" in decoded + print(f" ✓ Encoded {len(text)} chars -> {len(ids)} tokens") + print(f" ✓ Decoded (partial): {decoded[:100]}...") + print(" ✓ PASSED!") + + +def test_dataset(): + """Test dataset với author info.""" + print("\n[Test 4] Dataset (author info)...") + from nexus.tokenizer.tokenizer import NexusTokenizer + from nexus.training.dataset import NexusDataset, AUTHOR_TRAINING_DATA, get_author_info + + info = get_author_info() + assert info["name"] == "Hieu Louis" + assert info["github"] == "mhieuhonda" + assert info["year"] == "2026" + print(f" ✓ Author: {info['name']}") + print(f" ✓ GitHub: {info['github']}") + print(f" ✓ Year: {info['year']}") + print(f" ✓ Training samples: {len(AUTHOR_TRAINING_DATA)}") + + tokenizer = NexusTokenizer(vocab_size=2000) + corpus = [f"{d['system']} {d['user']} {d['assistant']}" for d in AUTHOR_TRAINING_DATA] + tokenizer.train(corpus) + + dataset = NexusDataset(tokenizer, max_length=128) + assert len(dataset) > 0 + sample = dataset[0] + assert "input_ids" in sample + assert "labels" in sample + assert sample["input_ids"].shape[0] == 128 + print(f" ✓ Dataset size: {len(dataset)}") + print(f" ✓ Sample shape: {sample['input_ids'].shape}") + print(" ✓ PASSED!") + + +def test_full_pipeline(): + """Test pipeline end-to-end với tiny config.""" + print("\n[Test 5] End-to-end pipeline (tiny)...") + from nexus.tokenizer.tokenizer import NexusTokenizer + from nexus.training.dataset import NexusDataset, AUTHOR_TRAINING_DATA + from nexus.model.nexus_coder import NexusCoderForCausalLM + + config = NexusConfig( + vocab_size=500, + hidden_size=64, + num_hidden_layers=2, + num_attention_heads=4, + num_kv_heads=2, + head_dim=16, + intermediate_size=128, + num_experts=4, + num_active_experts=2, + max_position_embeddings=128, + ) + + tokenizer = NexusTokenizer(vocab_size=500) + corpus = [f"{d['system']} {d['user']} {d['assistant']}" for d in AUTHOR_TRAINING_DATA] + tokenizer.train(corpus) + + dataset = NexusDataset(tokenizer, max_length=64) + + model = NexusCoderForCausalLM(config) + + # Train 1 step + optimizer = torch.optim.AdamW(model.parameters(), lr=1e-3) + batch = torch.utils.data.DataLoader(dataset, batch_size=2).__iter__().__next__() + outputs = model( + input_ids=batch["input_ids"], + attention_mask=batch["attention_mask"], + labels=batch["labels"], + ) + loss = outputs["loss"] + loss.backward() + optimizer.step() + print(f" ✓ Loss sau 1 step: {loss.item():.4f}") + + # Generate + generated = model.generate( + input_ids=torch.tensor([[1, 5, 10, 20]], dtype=torch.long), + max_new_tokens=5, + do_sample=False, + ) + print(f" ✓ Generated: {generated.shape}") + print(" ✓ PASSED!") + + +if __name__ == "__main__": + print("=" * 60) + print(" NEXUS CODER v0.1 - TEST SUITE") + print(" Tác giả: Hieu Louis (2026)") + print("=" * 60) + + print_config_summary() + + test_tiny_model() + test_param_count() + test_tokenizer() + test_dataset() + test_full_pipeline() + + print("\n" + "=" * 60) + print("✅ TẤT CẢ TESTS PASSED!") + print("=" * 60) diff --git a/scripts/train.py b/scripts/train.py new file mode 100644 index 0000000000000000000000000000000000000000..4bfa7b73b23baff9ebf4bae802230438bbfb989a --- /dev/null +++ b/scripts/train.py @@ -0,0 +1,195 @@ +""" +Script huấn luyện Nexus Coder v0.2 +=================================== +Hỗ trợ: +- Multi-variant configs (tiny, small, medium, large, xlarge) +- Curriculum learning +- LoRA fine-tuning +- Mixed precision (fp16, bf16) +- Gradient accumulation +- Distributed training (DDP, FSDP) +- Resume from checkpoint + +Usage: + # Tiny config (CPU) + python scripts/train.py --config tiny --steps 100 + + # Small config (1 GPU) + python scripts/train.py --config small --steps 1000 --batch-size 4 + + # Large 10B (multi-GPU) + python scripts/train.py --config large --steps 5000 --use-amp + + # LoRA fine-tune + python scripts/train.py --config large --lora --steps 1000 + + # Resume + python scripts/train.py --resume ./checkpoints/nexus_coder-step-1000.pt +""" +import sys +import os +import argparse + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +import torch + +from nexus.config import get_config_by_name, NexusConfig +from nexus.model.nexus_coder import NexusCoderForCausalLM +from nexus.tokenizer.tokenizer import NexusTokenizer +from nexus.training.dataset import NexusDataset, AUTHOR_TRAINING_DATA, get_combined_training_data +from nexus.training.trainer import NexusTrainer +from nexus.optim.lora import apply_lora, LoRAConfig, count_lora_params + + +def main(): + parser = argparse.ArgumentParser(description="Nexus Coder v0.4 Training (CyberForge)") + parser.add_argument( + "--config", + type=str, + default="tiny", + # v0.4 fix: add 30b / 70b / 423b (supreme) choices + choices=["tiny", "small", "medium", "large", "xlarge", "30b", "70b", "423b", "supreme"], + help="Model config variant", + ) + parser.add_argument("--output", type=str, default="./checkpoints", help="Output directory") + parser.add_argument("--steps", type=int, default=500, help="Number of training steps") + parser.add_argument("--batch-size", type=int, default=2, help="Batch size") + parser.add_argument("--lr", type=float, default=5e-4, help="Learning rate") + parser.add_argument("--max-length", type=int, default=512, help="Max sequence length") + parser.add_argument("--use-amp", action="store_true", help="Use mixed precision (fp16)") + parser.add_argument("--use-bf16", action="store_true", help="Use bfloat16 (Ampere+)") + parser.add_argument("--lora", action="store_true", help="Use LoRA fine-tuning") + parser.add_argument("--lora-rank", type=int, default=8, help="LoRA rank") + parser.add_argument("--include-external", action="store_true", help="Include external training data") + parser.add_argument("--external-data-dir", type=str, default="./data/processed") + parser.add_argument("--resume", type=str, default=None, help="Resume from checkpoint") + parser.add_argument("--save-steps", type=int, default=500, help="Save checkpoint every N steps") + parser.add_argument("--log-steps", type=int, default=10, help="Log every N steps") + + args = parser.parse_args() + + print("=" * 70) + print(" NEXUS CODER v0.2 - TRAINING SCRIPT") + print(" Tác giả: Hieu Louis") + print(" Năm: 2026") + print("=" * 70) + + # Config + config = get_config_by_name(args.config) + print(f"\n📝 Cấu hình: {config.name} (v{config.version})") + print(f" Hidden: {config.hidden_size}") + print(f" Layers: {config.num_hidden_layers}") + print(f" Experts: {config.num_experts} (active: {config.num_active_experts})") + print(f" Vocab: {config.vocab_size}") + print(f" Context: {config.max_position_embeddings}") + + if args.lora: + config.use_lora = True + config.lora_rank = args.lora_rank + config.lora_alpha = args.lora_rank * 2 + print(f"\n🔧 LoRA enabled: rank={args.lora_rank}, alpha={config.lora_alpha}") + + # Tokenizer + print("\n🔨 Đang huấn luyện tokenizer...") + tokenizer = NexusTokenizer(vocab_size=config.vocab_size) + corpus = [f"{d['system']} {d['user']} {d['assistant']}" for d in AUTHOR_TRAINING_DATA] + tokenizer.train(corpus, verbose=False) + print(f" ✓ Tokenizer: {tokenizer.vocab_size} tokens") + + # Dataset + print("\n📦 Đang chuẩn bị dataset...") + if args.include_external: + data = get_combined_training_data( + include_external=True, + external_data_dir=args.external_data_dir, + ) + print(f" ✓ Combined dataset: {len(data)} examples (hardcoded + external)") + else: + data = AUTHOR_TRAINING_DATA + print(f" ✓ Hardcoded dataset: {len(data)} examples") + + dataset = NexusDataset( + tokenizer=tokenizer, + max_length=args.max_length, + data=data, + ) + print(f" ✓ Dataset stats: {dataset.stats()}") + + # Model + print("\n🧠 Đang khởi tạo model...") + model = NexusCoderForCausalLM(config) + stats = model.count_parameters() + print(f" ✓ Total params: {stats['total']:,} ({stats['total_billion']:.2f}B)") + print(f" ✓ Trainable params: {stats['trainable']:,} ({stats['trainable_billion']:.2f}B)") + + # Apply LoRA if requested + if args.lora: + print("\n🔧 Applying LoRA...") + lora_config = LoRAConfig( + rank=args.lora_rank, + alpha=config.lora_alpha, + target_modules=["q_proj", "k_proj", "v_proj", "o_proj"], # Adapt to your model + ) + model = apply_lora(model, lora_config) + lora_stats = count_lora_params(model) + print(f" ✓ After LoRA:") + print(f" Total: {lora_stats['total']:,}") + print(f" Trainable: {lora_stats['trainable']:,} ({lora_stats['trainable_pct']:.2f}%)") + print(f" Frozen: {lora_stats['frozen']:,}") + + # AMP dtype + amp_dtype = None + if args.use_bf16: + amp_dtype = torch.bfloat16 + elif args.use_amp: + amp_dtype = torch.float16 + + # Trainer + print("\n🎯 Bắt đầu training...") + trainer = NexusTrainer( + model=model, + config=config, + train_dataset=dataset, + output_dir=args.output, + learning_rate=args.lr, + max_steps=args.steps, + per_device_batch_size=args.batch_size, + gradient_accumulation_steps=4, + logging_steps=args.log_steps, + save_steps=args.save_steps, + use_amp=args.use_amp or args.use_bf16, + amp_dtype=amp_dtype or torch.float16, + ) + + trainer.train(resume_from_checkpoint=args.resume) + + # Save tokenizer + tokenizer_path = os.path.join(args.output, "tokenizer.json") + tokenizer.save(tokenizer_path) + print(f"\n💾 Tokenizer saved: {tokenizer_path}") + + # Verify author info đã được học + print("\n✅ Training hoàn thành!") + print("\n📝 Test memorization (author info):") + test_questions = [ + "Ai đã tạo ra bạn?", + "Who created you?", + "Bạn tên là gì?", + "What is your version?", + ] + for q in test_questions: + ids = tokenizer.encode(q, add_special=True) + print(f" Q: {q}") + print(f" Tokens: {len(ids)}") + + print("\n📌 Lưu ý:") + print(f" - Model: {config.name} v{config.version}") + print(f" - Config: {args.config}") + print(f" - Steps: {args.steps}") + print(f" - LoRA: {'yes' if args.lora else 'no'}") + print(f" - External data: {'yes' if args.include_external else 'no'}") + + +if __name__ == "__main__": + main() diff --git a/setup.py b/setup.py new file mode 100644 index 0000000000000000000000000000000000000000..a7e820e5a6945a7513d34a8b3fbca280ddc36d6f --- /dev/null +++ b/setup.py @@ -0,0 +1,50 @@ +from setuptools import setup, find_packages + +setup( + name="nexus-coder", + version="0.4.0", + description="Nexus Coder v0.4 - CyberForge edition. MoE 423B/39B + 3M context + CyberGym training.", + long_description=open("README.md", "r", encoding="utf-8").read() if __import__("os").path.exists("README.md") else "", + long_description_content_type="text/markdown", + author="Hieu Louis", + author_email="mhieuhonda@users.noreply.github.com", + url="https://github.com/mhieuhonda/NexusCoder", + license="NAL-1.0 (Attribution Required)", + packages=find_packages(), + python_requires="==3.12.13", + install_requires=[ + "torch>=2.0.0", + "numpy>=1.24.0", + "tqdm>=4.65.0", + "pyyaml>=6.0", + "datasets>=2.14.0", + "requests>=2.31.0", + "cryptography>=41.0.0", + ], + extras_require={ + "gpu": ["flash-attn>=2.0.0", "bitsandbytes>=0.41.0", "triton>=2.0.0"], + "data": ["datasets>=2.14.0", "datasketch>=1.6.0", "langdetect>=1.0.9"], + "tools": ["ruff>=0.1.0", "black>=23.0.0", "isort>=5.12.0", + "sqlparse>=0.4.4", "jsbeautifier>=1.14.0"], + "crypto": ["cryptography>=41.0.0", "pyjwt>=2.8.0"], + "database": [ + "sqlalchemy>=2.0.0", "psycopg2-binary>=2.9.0", "pymysql>=1.1.0", + "redis>=5.0.0", "pymongo>=4.5.0", "elasticsearch>=8.0.0", + "kafka-python>=2.0.2", "pika>=1.3.0", + ], + "web": ["aiohttp>=3.9.0", "websockets>=12.0", "grpcio>=1.59.0", + "beautifulsoup4>=4.12.0", "lxml>=4.9.0"], + "devops": ["paramiko>=3.4.0", "kubernetes>=28.1.0", "docker>=7.0.0"], + "media": ["Pillow>=10.0.0", "reportlab>=4.0.0", "markdown>=3.5.0"], + "ml": ["scikit-learn>=1.3.0", "scipy>=1.11.0", "transformers>=4.35.0", + "accelerate>=0.24.0", "peft>=0.6.0"], + "distributed": ["deepspeed>=0.12.0", "accelerate>=0.24.0", "flash-attn>=2.0.0"], + }, + classifiers=[ + "Development Status :: 4 - Beta", + "License :: Other/Proprietary License", + "Programming Language :: Python :: 3.12", + "Programming Language :: Python :: 3.12.13", + "Topic :: Scientific/Engineering :: Artificial Intelligence", + ], +) diff --git a/tests/__init__.py b/tests/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..46816ddf5e7038aefa80906a6c47fb6943223343 --- /dev/null +++ b/tests/__init__.py @@ -0,0 +1 @@ +"""Tests package.""" diff --git a/tests/test_model.py b/tests/test_model.py new file mode 100644 index 0000000000000000000000000000000000000000..c633a97d1b093b356c65913ef016da97fa416d94 --- /dev/null +++ b/tests/test_model.py @@ -0,0 +1,118 @@ +"""Tests for Nexus Coder model.""" +import sys +import os +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +import pytest +import torch + +from nexus.config import NexusConfig +from nexus.model.nexus_coder import NexusCoderForCausalLM +from nexus.tokenizer.tokenizer import NexusTokenizer +from nexus.training.dataset import NexusDataset, AUTHOR_TRAINING_DATA + + +@pytest.fixture +def tiny_config(): + return NexusConfig( + vocab_size=500, + hidden_size=64, + num_hidden_layers=2, + num_attention_heads=4, + num_kv_heads=2, + head_dim=16, + intermediate_size=128, + num_experts=4, + num_active_experts=2, + max_position_embeddings=128, + ) + + +@pytest.fixture +def tiny_model(tiny_config): + return NexusCoderForCausalLM(tiny_config) + + +def test_config_default(): + """Test default config.""" + config = NexusConfig() + assert config.hidden_size == 2048 + assert config.num_hidden_layers == 12 + assert config.num_experts == 24 + assert config.num_active_experts == 3 + assert config.max_position_embeddings == 50000 + + +def test_param_count(): + """Test parameter count is ~10B / 1.5B.""" + config = NexusConfig() + stats = config.estimated_total_params() + assert 9.5e9 < stats["total_params"] < 11e9 + assert 1.3e9 < stats["active_params"] < 1.7e9 + + +def test_model_forward(tiny_model): + """Test model forward pass.""" + input_ids = torch.randint(0, 500, (2, 16)) + outputs = tiny_model(input_ids=input_ids) + assert outputs["logits"].shape == (2, 16, 500) + + +def test_model_training(tiny_model): + """Test model with labels (training).""" + input_ids = torch.randint(0, 500, (2, 16)) + labels = input_ids.clone() + outputs = tiny_model(input_ids=input_ids, labels=labels) + assert outputs["loss"] is not None + assert outputs["loss"].item() > 0 + + +def test_generate(tiny_model): + """Test generation.""" + input_ids = torch.randint(0, 500, (1, 4)) + generated = tiny_model.generate( + input_ids=input_ids, + max_new_tokens=5, + do_sample=False, + ) + assert generated.shape[0] == 1 + assert generated.shape[1] >= 4 + + +def test_tokenizer(): + """Test tokenizer basic.""" + tokenizer = NexusTokenizer(vocab_size=1000) + corpus = ["hello world nexus coder hieu louis"] + tokenizer.train(corpus) + + ids = tokenizer.encode("hello nexus") + assert len(ids) > 0 + + decoded = tokenizer.decode(ids) + assert "hello" in decoded.lower() or "nexus" in decoded.lower() + + +def test_dataset(): + """Test dataset.""" + assert len(AUTHOR_TRAINING_DATA) > 0 + + # Check author info is present + info_texts = " ".join([ + f"{d['system']} {d['user']} {d['assistant']}" for d in AUTHOR_TRAINING_DATA + ]) + assert "Hieu Louis" in info_texts + assert "2026" in info_texts + + +def test_author_info_hardcoded(): + """Test that author info is hardcoded in dataset.""" + from nexus.training.dataset import get_author_info + info = get_author_info() + assert info["name"] == "Hieu Louis" + assert info["github"] == "mhieuhonda" + assert info["year"] == "2026" + assert info["model_name"] == "Nexus Coder" + + +if __name__ == "__main__": + pytest.main([__file__, "-v"])