# Piko-9b task runner. # # PIKO_MODEL_PATH selects the checkpoint for every target that needs weights. # Point it at a local directory on fast internal storage — loading 21 GB from an # external USB disk takes 10-20 minutes per invocation instead of ~100 seconds. PYTHON ?= python PIKO_MODEL ?= Dexy2/Piko-9b BASE_MODEL ?= Qwen/Qwen3.5-9B QUANT ?= 4bit RESULTS := evaluation/results BENCH_RESULTS := benchmarks/results .DEFAULT_GOAL := help .PHONY: help install install-dev format lint test test-all smoke-eval custom-eval \ baseline-eval compare benchmark profile audit lineage validate-model-card \ assets clean check help: ## Show this help @grep -hE '^[a-zA-Z_-]+:.*?## ' $(MAKEFILE_LIST) \ | awk 'BEGIN{FS=":.*?## "}{printf " \033[36m%-20s\033[0m %s\n", $$1, $$2}' install: ## Install runtime dependencies $(PYTHON) -m pip install -r requirements.txt install-dev: ## Install runtime + evaluation + dev dependencies $(PYTHON) -m pip install -r requirements-dev.txt pre-commit install format: ## Format code ruff format examples scripts tests evaluation benchmarks lint: ## Lint code and check formatting ruff check examples scripts tests evaluation benchmarks ruff format --check examples scripts tests evaluation benchmarks test: ## Fast tests only — no weights, no GPU, safe for CI $(PYTHON) -m pytest tests -v -m "not slow" test-all: ## Every test, including those that load the checkpoint @test -n "$(PIKO_MODEL_PATH)" || \ { echo "PIKO_MODEL_PATH must point at a local checkpoint"; exit 1; } PIKO_MODEL_PATH=$(PIKO_MODEL_PATH) $(PYTHON) -m pytest tests -v assets: ## Render the synthetic evaluation fixtures $(PYTHON) evaluation/custom_suite/build_assets.py smoke-eval: assets ## ~5 minute sanity check (catches degenerate output first) $(PYTHON) evaluation/run_smoke_eval.py \ --model $(PIKO_MODEL) --quantization $(QUANT) \ --output $(RESULTS)/smoke_piko-9b.json custom-eval: assets ## Full 70-case regression suite against Piko-9b $(PYTHON) evaluation/custom_suite/run_custom_eval.py \ --model $(PIKO_MODEL) --label piko-9b --quantization $(QUANT) \ --max-new-tokens 512 --output $(RESULTS)/custom_suite_piko-9b.json baseline-eval: assets ## Same suite against the base model — required for any comparison $(PYTHON) evaluation/custom_suite/run_custom_eval.py \ --model $(BASE_MODEL) --label qwen3.5-9b-base --quantization $(QUANT) \ --max-new-tokens 512 --output $(RESULTS)/custom_suite_qwen3.5-9b-base.json compare: ## Build the side-by-side table (refuses mismatched run settings) $(PYTHON) evaluation/compare_results.py \ --candidate $(RESULTS)/custom_suite_piko-9b.json \ --baseline $(RESULTS)/custom_suite_qwen3.5-9b-base.json \ --output $(RESULTS)/comparison.md benchmark: custom-eval baseline-eval compare ## Candidate + baseline + comparison profile: ## Throughput, latency and memory profiling $(PYTHON) benchmarks/profile_inference.py \ --model $(PIKO_MODEL) --quantization $(QUANT) \ --image evaluation/custom_suite/assets/receipt.png \ --output $(BENCH_RESULTS)/inference_$(QUANT).json $(PYTHON) benchmarks/profile_memory.py \ --model $(PIKO_MODEL) --quantization $(QUANT) \ --output $(BENCH_RESULTS)/memory_$(QUANT).json audit: ## Regenerate the repository audit @test -n "$(PIKO_MODEL_PATH)" || \ { echo "PIKO_MODEL_PATH must point at a local checkpoint"; exit 1; } $(PYTHON) scripts/audit_repository.py \ --model $(PIKO_MODEL_PATH) --output reports/repository_audit.json lineage: ## Re-verify lineage by tensor comparison (needs all three checkpoints) @test -n "$(LANGUAGE_CKPT)" -a -n "$(VISION_CKPT)" || \ { echo "Set LANGUAGE_CKPT and VISION_CKPT"; exit 1; } $(PYTHON) scripts/analyze_lineage.py \ --candidate $(PIKO_MODEL_PATH) --language $(LANGUAGE_CKPT) --vision $(VISION_CKPT) \ --output reports/lineage_analysis.json validate-model-card: ## Check the model card's metadata, links and claims $(PYTHON) scripts/validate_model_card.py --readme README.md check: lint test validate-model-card ## Everything that runs without weights clean: ## Remove caches and generated fixtures rm -rf .pytest_cache .ruff_cache **/__pycache__ rm -f evaluation/custom_suite/assets/*.png evaluation/prompts/assets/*.png