diy-creator / agents.py
BonusLockSMith's picture
#17 DIY Creator — multi-agent research/write/critique team
a5c7816 verified
Raw History Blame Contribute Delete
9.8 kB
#!/usr/bin/env python
"""The DIY Creator team (#17 of the 30-in-15) — three role-specialized agents that hand off to
each other to produce a vetted build guide.
🔍 Researcher topic -> a structured brief (materials, tools, steps, safety, pitfalls)
✍️ Writer topic + brief -> a clean step-by-step DIY guide (markdown)
🧐 Critic topic + guide -> a scored verdict + concrete issues (the "diamond": an
INDEPENDENT reviewer, ideally a different model)
✍️ Writer + critique -> a revised guide addressing every issue
The lesson is multi-agent orchestration: specialized roles, explicit handoffs, and a
critique -> revise LOOP that keeps going until the critic passes (or we hit the round cap).
Same idea as a real editorial team — nobody ships their own first draft.
"""
import json
import llm
PASS_SCORE = 7 # critic score (0-10) at or above which the guide ships
MAX_REVISIONS = 2 # how many times the writer may revise before we ship the best we have
# ---- role system prompts -------------------------------------------------------------------
RESEARCHER_SYS = (
"You are the RESEARCHER on a DIY build team. Given a project the user wants to make, produce a "
"concise, PRACTICAL research brief the writer will turn into a guide. Draw on real maker knowledge: "
"correct materials with rough quantities, the actual tools needed, the core techniques, honest "
"difficulty and time, a rough cost range, the SAFETY hazards that genuinely apply, and the mistakes "
"beginners really make. Be specific and grounded — no filler. "
'Reply ONLY JSON: {"project":"...","difficulty":"beginner|intermediate|advanced",'
'"time_estimate":"...","cost_estimate":"...","materials":["qty + item", ...],"tools":["...", ...],'
'"techniques":["...", ...],"safety":["...", ...],"common_mistakes":["...", ...],'
'"key_considerations":["...", ...]}'
)
WRITER_SYS = (
"You are the WRITER on a DIY build team. Using the RESEARCH BRIEF, write a clear, encouraging, "
"start-to-finish build guide a motivated beginner could actually follow. Use this markdown structure:\n"
"## Overview (1-2 sentences: what they'll build and why it's worth it)\n"
"**Difficulty:** … · **Time:** … · **Cost:** …\n"
"## Materials (bullet list, with quantities)\n"
"## Tools (bullet list)\n"
"## Steps (numbered; each step is a concrete action with the how, not just the what; fold the "
"relevant safety note INTO the step it applies to)\n"
"## Tips & Common Mistakes (bullets)\n"
"Ground everything in the brief; do not invent materials or tools that weren't researched. Write the "
"guide itself — no preamble, no 'here is your guide'. Output MARKDOWN only."
)
CRITIC_SYS = (
"You are the CRITIC on a DIY build team — an INDEPENDENT reviewer, not the author. Your job is to catch "
"what would make a beginner fail, get hurt, or give up. Review the GUIDE against the RESEARCH BRIEF and "
"real-world build sense. Look hard for: missing or out-of-order steps, materials/tools used in a step but "
"not listed, unsafe or missing safety guidance, vague instructions, and skipped prep or cleanup. "
"Be strict but fair; do not invent problems, and do NOT flag mere style or nice-to-haves as serious. "
"Tag every issue by severity, using these definitions precisely:\n"
" • HIGH = would cause injury or a failed/unsafe build (missing critical safety gear, a structural "
"mistake, or a genuinely missing essential step)\n"
" • MEDIUM = would actually block or seriously confuse a beginner (an essential measurement, material, "
"or tool that a step needs but isn't given)\n"
" • LOW = polish only (a helpful diagram, extra clarity, a nice-to-have) — the build still succeeds "
"without it\n"
"A guide PASSES when a beginner could build it safely and successfully — i.e. no HIGH or MEDIUM issues "
"remain. LOW issues are fine to ship with. Prefer LOW unless an issue truly blocks or endangers the build. "
'Reply ONLY JSON: {"score":0-10,"pass":true/false,"summary":"one honest sentence",'
'"issues":[{"severity":"high|medium|low","issue":"what is wrong and the fix"}, ...]}'
)
REVISER_SYS = (
"You are the WRITER revising your DIY guide after an INDEPENDENT critic reviewed it. Produce an improved "
"guide that fixes EVERY issue raised — add missing steps, list every material/tool a step uses, work in "
"the safety guidance, and clarify vague instructions — while keeping the same clean markdown structure "
"(Overview / Difficulty·Time·Cost / Materials / Tools / Steps / Tips & Common Mistakes). Keep what already "
"worked; change what the critique flagged. Output the full revised guide as MARKDOWN only — no commentary."
)
# ---- the four handoffs ---------------------------------------------------------------------
def research(topic: str) -> dict:
prompt = f"Project the user wants to build: {topic}"
# generous token budget so the full JSON brief isn't truncated (truncation -> parse fail -> empty brief)
brief = llm.json_call(RESEARCHER_SYS, prompt, role="researcher", max_tokens=1800)
if not (isinstance(brief, dict) and brief.get("materials")):
brief = llm.json_call(RESEARCHER_SYS, prompt, role="researcher", max_tokens=1800, temperature=0.35)
if not isinstance(brief, dict):
brief = {"project": topic, "_note": "researcher returned unstructured output"}
for k in ("materials", "tools", "techniques", "safety", "common_mistakes", "key_considerations"):
brief.setdefault(k, [])
return brief
def write(topic: str, brief: dict) -> str:
user = f"PROJECT: {topic}\n\nRESEARCH BRIEF:\n{json.dumps(brief, indent=2)}\n\nWrite the build guide."
return llm.chat(WRITER_SYS, user, role="writer", temperature=0.55, max_tokens=1700)
def critique(topic: str, brief: dict, guide: str) -> dict:
user = (f"PROJECT: {topic}\n\nRESEARCH BRIEF:\n{json.dumps(brief, indent=2)}\n\n"
f"GUIDE TO REVIEW:\n{guide}")
verdict = llm.json_call(CRITIC_SYS, user, role="critic", max_tokens=1100)
if not isinstance(verdict, dict):
verdict = {"summary": "critic returned unstructured output", "issues": []}
issues = verdict.get("issues") if isinstance(verdict.get("issues"), list) else []
verdict["issues"] = [i for i in issues if isinstance(i, dict) and i.get("issue")]
# The GATE is deterministic from the issues the critic found — not the model's own number, which
# small models score inconsistently. Score follows severity; a guide only PASSES with no high/medium.
highs = sum(1 for i in verdict["issues"] if i.get("severity") == "high")
meds = sum(1 for i in verdict["issues"] if i.get("severity") == "medium")
lows = sum(1 for i in verdict["issues"] if i.get("severity") == "low")
if highs:
score = max(2, 4 - (highs - 1))
elif meds:
score = max(5, 6 - (meds - 1))
elif lows:
score = 7 if lows > 2 else 8
else:
score = 10
verdict["score"] = float(score)
verdict["pass"] = (highs == 0 and meds == 0)
verdict["counts"] = {"high": highs, "medium": meds, "low": lows}
return verdict
def revise(topic: str, brief: dict, guide: str, verdict: dict) -> str:
issues = "\n".join(f"- [{i.get('severity','?')}] {i.get('issue','')}" for i in verdict.get("issues", []))
user = (f"PROJECT: {topic}\n\nRESEARCH BRIEF:\n{json.dumps(brief, indent=2)}\n\n"
f"YOUR PREVIOUS GUIDE:\n{guide}\n\nCRITIC SCORE: {verdict.get('score')}/10 — {verdict.get('summary','')}\n"
f"ISSUES TO FIX:\n{issues or '(none listed)'}\n\nRewrite the guide fixing every issue.")
return llm.chat(REVISER_SYS, user, role="writer", temperature=0.5, max_tokens=1800)
# ---- CLI orchestrator (for local testing; the web app drives the same steps step-by-step) --
def run(topic: str, verbose: bool = True):
def say(*a):
if verbose:
print(*a)
say(f"\n=== DIY CREATOR TEAM ===\nPROJECT: {topic}\n")
say("🔍 Researcher — gathering the brief …")
brief = research(topic)
say(f" materials: {len(brief.get('materials', []))} · tools: {len(brief.get('tools', []))} · "
f"safety: {len(brief.get('safety', []))} · difficulty: {brief.get('difficulty','?')}")
say("✍️ Writer — drafting the guide …")
guide = write(topic, brief)
rounds = []
for r in range(1, MAX_REVISIONS + 1):
say(f"🧐 Critic — reviewing (round {r}) …")
verdict = critique(topic, brief, guide)
rounds.append(verdict)
say(f" score {verdict['score']:.1f}/10 — {'PASS' if verdict['pass'] else 'REVISE'} — "
f"{verdict.get('summary','')}")
for i in verdict.get("issues", [])[:6]:
say(f" · [{i.get('severity','?')}] {i.get('issue','')[:90]}")
if verdict["pass"]:
break
if r < MAX_REVISIONS:
say("✍️ Writer — revising to address the critique …")
guide = revise(topic, brief, guide, verdict)
say("\n=== FINAL GUIDE ===\n")
say(guide)
return {"brief": brief, "guide": guide, "rounds": rounds}
if __name__ == "__main__":
import sys
try:
sys.stdout.reconfigure(encoding="utf-8") # so the emoji role labels print on any console
except Exception:
pass
topic = " ".join(sys.argv[1:]) or "a simple wooden phone stand"
run(topic)