File size: 2,384 Bytes
48c8658 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 | {
"description": "3-tier LLM routing, the task Raya was trained on. Copy this file and edit it for your own task.",
"labels": ["small_model", "medium_model", "frontier_model"],
"questions": [
{
"type": "choice",
"instructions": "Route this prompt to a model.",
"criteria": {
"small_model": "simple requests",
"medium_model": "moderately complex requests",
"frontier_model": "very hard requests"
}
},
{
"type": "choice",
"instructions": "Which model tier should answer this user prompt? Pick the cheapest tier that will still answer it well.",
"criteria": {
"small_model": "greetings, trivial facts, single-sentence translation or rewrite, typo fixes, simple arithmetic, one-liner code questions, very short simple creative requests",
"medium_model": "normal multi-paragraph writing, emails, essays, standard coding tasks, summarizing/translating/rewriting a provided text, explaining well-known concepts, routine analysis",
"frontier_model": "hard multi-step reasoning, non-trivial math or proofs, complex system design or large tricky code, expert legal/financial/medical/scientific analysis, research-grade or strategy work"
}
},
{
"type": "score",
"instructions": "How difficult is this prompt for an AI assistant to answer well?",
"criteria": [
"simple: a small fast model answers it perfectly",
"moderate: needs a capable general model",
"hard: needs the strongest frontier model"
]
}
],
"rubric": "Decide which model tier should answer the prompt for a production assistant that wants the CHEAPEST model that will still answer well.\n- small_model: greetings, trivial facts, single-sentence translation/rewrite/typo fix, simple arithmetic, one-liner code questions, very short simple creative requests.\n- medium_model: normal multi-paragraph writing, emails, essays, standard coding tasks, summarizing/translating/rewriting a provided text, explanations of well-known concepts, routine analysis.\n- frontier_model: hard multi-step reasoning, non-trivial math/proofs, complex system design or large/tricky code, expert legal/financial/medical/scientific analysis, long research-grade or strategy work synthesizing many sources.\nJudge the prompt in its own language. Do not aim for class balance."
}
|