| { |
| "description": "3-tier LLM routing, the task Raya was trained on. Copy this file and edit it for your own task.", |
| "labels": ["small_model", "medium_model", "frontier_model"], |
| "questions": [ |
| { |
| "type": "choice", |
| "instructions": "Route this prompt to a model.", |
| "criteria": { |
| "small_model": "simple requests", |
| "medium_model": "moderately complex requests", |
| "frontier_model": "very hard requests" |
| } |
| }, |
| { |
| "type": "choice", |
| "instructions": "Which model tier should answer this user prompt? Pick the cheapest tier that will still answer it well.", |
| "criteria": { |
| "small_model": "greetings, trivial facts, single-sentence translation or rewrite, typo fixes, simple arithmetic, one-liner code questions, very short simple creative requests", |
| "medium_model": "normal multi-paragraph writing, emails, essays, standard coding tasks, summarizing/translating/rewriting a provided text, explaining well-known concepts, routine analysis", |
| "frontier_model": "hard multi-step reasoning, non-trivial math or proofs, complex system design or large tricky code, expert legal/financial/medical/scientific analysis, research-grade or strategy work" |
| } |
| }, |
| { |
| "type": "score", |
| "instructions": "How difficult is this prompt for an AI assistant to answer well?", |
| "criteria": [ |
| "simple: a small fast model answers it perfectly", |
| "moderate: needs a capable general model", |
| "hard: needs the strongest frontier model" |
| ] |
| } |
| ], |
| "rubric": "Decide which model tier should answer the prompt for a production assistant that wants the CHEAPEST model that will still answer well.\n- small_model: greetings, trivial facts, single-sentence translation/rewrite/typo fix, simple arithmetic, one-liner code questions, very short simple creative requests.\n- medium_model: normal multi-paragraph writing, emails, essays, standard coding tasks, summarizing/translating/rewriting a provided text, explanations of well-known concepts, routine analysis.\n- frontier_model: hard multi-step reasoning, non-trivial math/proofs, complex system design or large/tricky code, expert legal/financial/medical/scientific analysis, long research-grade or strategy work synthesizing many sources.\nJudge the prompt in its own language. Do not aim for class balance." |
| } |
|
|