#!/usr/bin/env python """The DIY Creator team (#17 of the 30-in-15) — three role-specialized agents that hand off to each other to produce a vetted build guide. 🔍 Researcher topic -> a structured brief (materials, tools, steps, safety, pitfalls) ✍️ Writer topic + brief -> a clean step-by-step DIY guide (markdown) 🧐 Critic topic + guide -> a scored verdict + concrete issues (the "diamond": an INDEPENDENT reviewer, ideally a different model) ✍️ Writer + critique -> a revised guide addressing every issue The lesson is multi-agent orchestration: specialized roles, explicit handoffs, and a critique -> revise LOOP that keeps going until the critic passes (or we hit the round cap). Same idea as a real editorial team — nobody ships their own first draft. """ import json import llm PASS_SCORE = 7 # critic score (0-10) at or above which the guide ships MAX_REVISIONS = 2 # how many times the writer may revise before we ship the best we have # ---- role system prompts ------------------------------------------------------------------- RESEARCHER_SYS = ( "You are the RESEARCHER on a DIY build team. Given a project the user wants to make, produce a " "concise, PRACTICAL research brief the writer will turn into a guide. Draw on real maker knowledge: " "correct materials with rough quantities, the actual tools needed, the core techniques, honest " "difficulty and time, a rough cost range, the SAFETY hazards that genuinely apply, and the mistakes " "beginners really make. Be specific and grounded — no filler. " 'Reply ONLY JSON: {"project":"...","difficulty":"beginner|intermediate|advanced",' '"time_estimate":"...","cost_estimate":"...","materials":["qty + item", ...],"tools":["...", ...],' '"techniques":["...", ...],"safety":["...", ...],"common_mistakes":["...", ...],' '"key_considerations":["...", ...]}' ) WRITER_SYS = ( "You are the WRITER on a DIY build team. Using the RESEARCH BRIEF, write a clear, encouraging, " "start-to-finish build guide a motivated beginner could actually follow. Use this markdown structure:\n" "## Overview (1-2 sentences: what they'll build and why it's worth it)\n" "**Difficulty:** … · **Time:** … · **Cost:** …\n" "## Materials (bullet list, with quantities)\n" "## Tools (bullet list)\n" "## Steps (numbered; each step is a concrete action with the how, not just the what; fold the " "relevant safety note INTO the step it applies to)\n" "## Tips & Common Mistakes (bullets)\n" "Ground everything in the brief; do not invent materials or tools that weren't researched. Write the " "guide itself — no preamble, no 'here is your guide'. Output MARKDOWN only." ) CRITIC_SYS = ( "You are the CRITIC on a DIY build team — an INDEPENDENT reviewer, not the author. Your job is to catch " "what would make a beginner fail, get hurt, or give up. Review the GUIDE against the RESEARCH BRIEF and " "real-world build sense. Look hard for: missing or out-of-order steps, materials/tools used in a step but " "not listed, unsafe or missing safety guidance, vague instructions, and skipped prep or cleanup. " "Be strict but fair; do not invent problems, and do NOT flag mere style or nice-to-haves as serious. " "Tag every issue by severity, using these definitions precisely:\n" " • HIGH = would cause injury or a failed/unsafe build (missing critical safety gear, a structural " "mistake, or a genuinely missing essential step)\n" " • MEDIUM = would actually block or seriously confuse a beginner (an essential measurement, material, " "or tool that a step needs but isn't given)\n" " • LOW = polish only (a helpful diagram, extra clarity, a nice-to-have) — the build still succeeds " "without it\n" "A guide PASSES when a beginner could build it safely and successfully — i.e. no HIGH or MEDIUM issues " "remain. LOW issues are fine to ship with. Prefer LOW unless an issue truly blocks or endangers the build. " 'Reply ONLY JSON: {"score":0-10,"pass":true/false,"summary":"one honest sentence",' '"issues":[{"severity":"high|medium|low","issue":"what is wrong and the fix"}, ...]}' ) REVISER_SYS = ( "You are the WRITER revising your DIY guide after an INDEPENDENT critic reviewed it. Produce an improved " "guide that fixes EVERY issue raised — add missing steps, list every material/tool a step uses, work in " "the safety guidance, and clarify vague instructions — while keeping the same clean markdown structure " "(Overview / Difficulty·Time·Cost / Materials / Tools / Steps / Tips & Common Mistakes). Keep what already " "worked; change what the critique flagged. Output the full revised guide as MARKDOWN only — no commentary." ) # ---- the four handoffs --------------------------------------------------------------------- def research(topic: str) -> dict: prompt = f"Project the user wants to build: {topic}" # generous token budget so the full JSON brief isn't truncated (truncation -> parse fail -> empty brief) brief = llm.json_call(RESEARCHER_SYS, prompt, role="researcher", max_tokens=1800) if not (isinstance(brief, dict) and brief.get("materials")): brief = llm.json_call(RESEARCHER_SYS, prompt, role="researcher", max_tokens=1800, temperature=0.35) if not isinstance(brief, dict): brief = {"project": topic, "_note": "researcher returned unstructured output"} for k in ("materials", "tools", "techniques", "safety", "common_mistakes", "key_considerations"): brief.setdefault(k, []) return brief def write(topic: str, brief: dict) -> str: user = f"PROJECT: {topic}\n\nRESEARCH BRIEF:\n{json.dumps(brief, indent=2)}\n\nWrite the build guide." return llm.chat(WRITER_SYS, user, role="writer", temperature=0.55, max_tokens=1700) def critique(topic: str, brief: dict, guide: str) -> dict: user = (f"PROJECT: {topic}\n\nRESEARCH BRIEF:\n{json.dumps(brief, indent=2)}\n\n" f"GUIDE TO REVIEW:\n{guide}") verdict = llm.json_call(CRITIC_SYS, user, role="critic", max_tokens=1100) if not isinstance(verdict, dict): verdict = {"summary": "critic returned unstructured output", "issues": []} issues = verdict.get("issues") if isinstance(verdict.get("issues"), list) else [] verdict["issues"] = [i for i in issues if isinstance(i, dict) and i.get("issue")] # The GATE is deterministic from the issues the critic found — not the model's own number, which # small models score inconsistently. Score follows severity; a guide only PASSES with no high/medium. highs = sum(1 for i in verdict["issues"] if i.get("severity") == "high") meds = sum(1 for i in verdict["issues"] if i.get("severity") == "medium") lows = sum(1 for i in verdict["issues"] if i.get("severity") == "low") if highs: score = max(2, 4 - (highs - 1)) elif meds: score = max(5, 6 - (meds - 1)) elif lows: score = 7 if lows > 2 else 8 else: score = 10 verdict["score"] = float(score) verdict["pass"] = (highs == 0 and meds == 0) verdict["counts"] = {"high": highs, "medium": meds, "low": lows} return verdict def revise(topic: str, brief: dict, guide: str, verdict: dict) -> str: issues = "\n".join(f"- [{i.get('severity','?')}] {i.get('issue','')}" for i in verdict.get("issues", [])) user = (f"PROJECT: {topic}\n\nRESEARCH BRIEF:\n{json.dumps(brief, indent=2)}\n\n" f"YOUR PREVIOUS GUIDE:\n{guide}\n\nCRITIC SCORE: {verdict.get('score')}/10 — {verdict.get('summary','')}\n" f"ISSUES TO FIX:\n{issues or '(none listed)'}\n\nRewrite the guide fixing every issue.") return llm.chat(REVISER_SYS, user, role="writer", temperature=0.5, max_tokens=1800) # ---- CLI orchestrator (for local testing; the web app drives the same steps step-by-step) -- def run(topic: str, verbose: bool = True): def say(*a): if verbose: print(*a) say(f"\n=== DIY CREATOR TEAM ===\nPROJECT: {topic}\n") say("🔍 Researcher — gathering the brief …") brief = research(topic) say(f" materials: {len(brief.get('materials', []))} · tools: {len(brief.get('tools', []))} · " f"safety: {len(brief.get('safety', []))} · difficulty: {brief.get('difficulty','?')}") say("✍️ Writer — drafting the guide …") guide = write(topic, brief) rounds = [] for r in range(1, MAX_REVISIONS + 1): say(f"🧐 Critic — reviewing (round {r}) …") verdict = critique(topic, brief, guide) rounds.append(verdict) say(f" score {verdict['score']:.1f}/10 — {'PASS' if verdict['pass'] else 'REVISE'} — " f"{verdict.get('summary','')}") for i in verdict.get("issues", [])[:6]: say(f" · [{i.get('severity','?')}] {i.get('issue','')[:90]}") if verdict["pass"]: break if r < MAX_REVISIONS: say("✍️ Writer — revising to address the critique …") guide = revise(topic, brief, guide, verdict) say("\n=== FINAL GUIDE ===\n") say(guide) return {"brief": brief, "guide": guide, "rounds": rounds} if __name__ == "__main__": import sys try: sys.stdout.reconfigure(encoding="utf-8") # so the emoji role labels print on any console except Exception: pass topic = " ".join(sys.argv[1:]) or "a simple wooden phone stand" run(topic)