{ "policy": "educational.policy", "version": 1, "subject": "duplicate a successful mindXtrain run", "exemplar": { "generation": 39, "model": "PYTHAI/mindXtrain39", "why_this_one": "the newest generation the imprint gate accepted; of the 37 attempts logged since, 30 proof_rejected (41, 46-74), 6 train_failed (40, 42-45, 75), none accepted" }, "claim": "A 135M model, two CPU cores and 70 minutes are enough to move proof-of-recall by +0.10. That is the whole of the claim: recall of a corpus, not identity and not reasoning.", "measured": { "_source": "gen39's own train.log (published at PYTHAI/mindXtrain39/train.log) and the ascent log entry for generation 39 — measured, not reconstructed", "steps": 116, "epochs": 2, "train_runtime_s": 4201, "ascent_wall_s": 4221.5, "train_loss": 1.65, "eval_loss": 1.225, "eval_entropy": 1.445, "eval_tokens": 348500, "train_examples_tokenized": 460, "s_per_step_mean": 36.2, "imprint": { "delta_recall": 0.1002, "imprinted": true, "stage": "accepted", "gate": "mindXtrain imprint (proof of recall)" } }, "recipe": { "_source": "the run recipe shape mindX writes per ascent (data/godel/ascend/genN/run.yaml); values here are the ones gen39's log confirms", "model": { "name": "HuggingFaceTB/SmolLM2-135M", "attn_implementation": "eager", "torch_dtype": "float32" }, "data": { "source": "mindx_dreams", "path": "data/memory/curated", "seq_len": 1024, "packing": true, "eval_split": 0.1, "include_evolutions": true, "max_samples": 1024 }, "train": { "backend": "trl_cpu", "method": { "kind": "lora", "r": 16, "alpha": 32, "dropout": 0.0, "target_modules": [ "q_proj", "k_proj", "v_proj", "o_proj" ] }, "optimizer": { "name": "adamw_torch", "lr": 0.0001 }, "schedule": { "type": "cosine", "warmup_ratio": 0.03, "epochs": 2 }, "batch": { "per_device": 1, "grad_accum": 8 }, "precision": "float32", "cpu_throttle": { "percent": 33, "nice": 19 } }, "hardware_measured_on": { "cpu_cores": 2, "cpu_model": "AMD EPYC 7543P", "ram_gb": 7.8, "gpu": null } }, "duplicate": { "framework": "https://github.com/professor-codephreak/mindXtrain", "steps": [ "uv sync --extra ml # trl + transformers + peft + accelerate", "mindxtrain init -t mindx_fallback_qwen3_1_5b_cpu_smoke -o run.yaml", "edit run.yaml to the recipe below (LoRA r16/α32 on q,k,v,o · lr 1e-4 cosine · 2 epochs · packing · seq 1024 · eval_split 0.1)", "mindxtrain train run.yaml --out out/runs --cpu-percent 33 --cpu-nice 19", "mindxtrain imprint --config run.yaml # the gate: recall BEFORE vs AFTER", "keep the run only if delta > the calibrated floor; otherwise it is a rejected generation and is recorded as one", "mindxtrain serve --config run.yaml --to ollama --tag # only after a positive imprint" ], "corpus": { "what": "mindX's curated machine.dream corpus (prose in the first person + Gödel decisions + persona rows)", "hub": "https://huggingface.co/datasets/PYTHAI/mindXascension/tree/main/machine.dream", "rule": "the corpus is rebuilt before every ascent; provenance per row in PROVENANCE.jsonl" } }, "gate": { "name": "imprint", "metric": "token-Jaccard recall of the corpus voice, after minus before", "floor": "calibrated by scripts/calibrate_min_delta.py (an untrained random-init adapter is the null); provisional 0.02", "decoding": { "do_sample": false, "repetition_penalty": 1.3, "no_repeat_ngram_size": 3 }, "honesty": "a positive imprint proves recall. It does not prove identity: the coach measured 16% identity on this lineage." }, "for_the_coach": { "read": [ "/insight/hf/coach", "/insight/hf/coach/results", "/insight/godel/ascend" ], "act": [ "POST /hf/spar/auto (score a generation)", "POST /hf/coach/recommend", "POST /hf/coach/recipe (adopt)" ], "rule": "the coach refuses a rung the ladder evidence already rejected" }, "orchestration": { "mastermind": "https://mastermind.pythai.net", "node": "https://mindx.pythai.net", "note": "Mastermind is the strategic layer that decides a campaign is worth running; mindXtrain is what runs it." } }