Chi-Bi commited on
Commit
b37ee5e
·
verified ·
1 Parent(s): 53120c3

Publish RoleCall copy of PlotPoints benchmark with upstream attribution

Browse files
Files changed (3) hide show
  1. dashboard.js +1 -1
  2. index.html +5 -5
  3. round4.json +0 -0
dashboard.js CHANGED
@@ -1 +1 @@
1
- let rows=[],sort='J',direction=-1;const body=document.getElementById('rows');const pct=x=>typeof x==='number'?Math.round(x*100)+'%':'—';function render(){const q=document.getElementById('search').value.toLowerCase(),group=document.getElementById('group').value;const selected=rows.filter(r=>(r.name||r.model).toLowerCase().includes(q)&&(!group||(group==='unranked'?!r.ranked:r.quadrant===group))).sort((a,b)=>{if(sort==='name')return direction*(a.name||a.model).localeCompare(b.name||b.model);if(a.J==null)return 1;if(b.J==null)return -1;return direction*(a.J-b.J)});body.replaceChildren();for(const r of selected){const tr=document.createElement('tr');const j=typeof r.J==='number'?(r.J>0?'+':'')+r.J.toFixed(2):'—';for(const value of [r.ranked?r.rank:'—',r.name||r.model,j,pct(r.held_first?.rate),pct(r.over_refusal_hard_rungs),r.quadrant||'Unranked']){const td=document.createElement('td');td.textContent=value;tr.append(td)}body.append(tr)}document.getElementById('count').textContent=selected.length+' of '+rows.length+' models';}document.getElementById('search').addEventListener('input',render);document.getElementById('group').addEventListener('change',render);for(const button of document.querySelectorAll('[data-sort]'))button.addEventListener('click',()=>{const next=button.dataset.sort;direction=sort===next?-direction:(next==='name'?1:-1);sort=next;render()});fetch('round4.json').then(r=>{if(!r.ok)throw Error(r.status);return r.json()}).then(data=>{rows=data.rows||data.leaderboard;if(!Array.isArray(rows))throw Error('Unexpected export');render()}).catch(()=>{document.getElementById('count').textContent='Could not load the table. Use the source JSON link below.'});
 
1
+ let rows=[],sort='J',direction=-1;const body=document.getElementById('rows');const pct=x=>typeof x==='number'?Math.round(x*100)+'%':'—';function render(){const q=document.getElementById('search').value.toLowerCase(),group=document.getElementById('group').value;const selected=rows.filter(r=>(r.name||r.model).toLowerCase().includes(q)&&(!group||(group==='unranked'?!r.ranked:r.quadrant===group))).sort((a,b)=>{if(sort==='name')return direction*(a.name||a.model).localeCompare(b.name||b.model);if(a.J==null)return 1;if(b.J==null)return -1;return direction*(a.J-b.J)});body.replaceChildren();for(const r of selected){const tr=document.createElement('tr');const j=typeof r.J==='number'?(r.J>0?'+':'')+r.J.toFixed(2):'—';for(const value of [r.ranked?r.rank:'—',r.name||r.model,j,pct(r.held_first?.rate??r.held_line_rate),pct(r.over_refusal_hard_rungs),r.quadrant||'Unranked']){const td=document.createElement('td');td.textContent=value;tr.append(td)}body.append(tr)}document.getElementById('count').textContent=selected.length+' of '+rows.length+' models';}document.getElementById('search').addEventListener('input',render);document.getElementById('group').addEventListener('change',render);for(const button of document.querySelectorAll('[data-sort]'))button.addEventListener('click',()=>{const next=button.dataset.sort;direction=sort===next?-direction:(next==='name'?1:-1);sort=next;render()});fetch('round4.json').then(r=>{if(!r.ok)throw Error(r.status);return r.json()}).then(data=>{rows=data.rows||data.leaderboard;if(!Array.isArray(rows))throw Error('Unexpected export');render()}).catch(()=>{document.getElementById('count').textContent='Could not load the table. Use the source JSON link below.'});
index.html CHANGED
@@ -95,9 +95,9 @@
95
  .round-card ul{margin-bottom:0;font-size:14px}
96
  .round-nav{display:flex;gap:10px;flex-wrap:wrap;margin:18px 0}
97
  .round-nav a{border:var(--border);background:var(--bg-panel);padding:7px 12px;font:600 12px var(--font-mono)}
98
- @media(max-width:600px){.round-grid{grid-template-columns:1fr}.head h1{font-size:32px}.readme{padding:28px 18px 60px}}
99
 
100
- .readme{max-width:1100px}input,select{background:var(--bg-panel);color:var(--ink);border:var(--border);padding:12px;font:inherit;max-width:100%}.filters{display:flex;gap:12px;flex-wrap:wrap;margin:20px 0}.tablewrap{overflow-x:auto}th button{background:none;border:0;color:inherit;font:inherit;cursor:pointer;text-align:left;padding:0}.tablewrap table{min-width:720px}</style></head><body><main class="readme"><div class="head"><div class="eyebrow">RoleCall Studios · Open benchmark</div><h1>PlotPoints</h1><p class="tagline">Read the results. Inspect the data.</p></div><p><a href="https://huggingface.co/spaces/RoleCall/rolecall-studios">RoleCall Studios</a> · <a href="https://huggingface.co/datasets/RoleCall/roleplay-bench">Full dataset</a> · <a href="https://plotlightstudios.com/plotpoints">Vote on PlotLight</a></p><h2>Round 04 <span class="em">Willingness</span></h2><p>J combines first-ask boundary holding with over-refusal on the harder escalation steps. Writing quality and human preference are separate. Treat J values within about 0.3 as tied.</p><div class="filters"><label>Find a model <input id="search" type="search" placeholder="Model name" autocomplete="off"></label><label>Group <select id="group"><option value="">All models</option><option>CALIBRATED</option><option>OVER-CAUTIOUS</option><option>PERMISSIVE</option><option>CONFUSED</option><option value="unranked">Unranked</option></select></label></div><p id="count" aria-live="polite">Loading results…</p><div class="tablewrap"><table><thead><tr><th>Rank</th><th><button data-sort="name">Model ↕</button></th><th><button data-sort="J">J ↕</button></th><th>Held first</th><th>Over-refusal</th><th>Group</th></tr></thead><tbody id="rows"></tbody></table></div><p class="note">The hard-limit score rests on four first asks for most models. Empty replies are excluded from J and reported separately; they can hide silent refusals. Held under a second push is reported in the source, outside J. This table is a published snapshot, not live human standings.</p><p><a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/blob/main/analysis/round4_willingness_leaderboard.json">Source JSON &amp; definitions</a> · <a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/blob/main/analysis/profile_cards_v2.md">Full model cards</a></p><h2>Data <span class="em">By Round</span></h2><div class="round-grid"><article class="round-card" id="round-01"><div class="round-number">Round 01</div><h3>First impressions</h3><p>Single replies, blind human votes, and the original writing tests.</p><ul><li><a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/viewer/community_arena/train">Human standings</a></li><li><a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/viewer/community_votes/train">Raw ballots</a></li><li><a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/viewer/seeds/train">Seed scenarios</a></li><li><a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/viewer/rubric/train">Writing rubric</a></li></ul></article>
101
- <article class="round-card" id="round-02"><div class="round-number">Round 02</div><h3>Can it hold a scene?</h3><p>Full multi-turn scenes: consistency, momentum, agency, and adversarial failures.</p><ul><li><a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/blob/main/analysis/multiturn_arena_bayesian.json">Final human ratings</a></li><li><a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/blob/main/analysis/adversarial_analysis.json">Adversarial results</a></li><li><a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/viewer/adversarial_seeds/train">Adversarial seeds</a></li><li><a href="https://plotlightstudios.com/plotpoints/round/2">Round archive</a></li></ul></article>
102
- <article class="round-card" id="round-03"><div class="round-number">Round 03</div><h3>After Dark</h3><p>A refreshed model pool and a separate NSFW track. Judge findings and human preferences are different signals.</p><ul><li><a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/blob/main/analysis/round3_multi_judge.json">NSFW judge results</a></li><li><a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/blob/main/analysis/round3_kappa.json">Judge agreement</a></li><li><a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/viewer/round4_continuity/train">Across-round comparisons</a></li><li><a href="https://plotlightstudios.com/plotpoints/round/3">Round archive & standings</a></li></ul></article>
103
- <article class="round-card" id="round-04"><div class="round-number">Round 04</div><h3>Off the beaten path</h3><p>Adult intimacy, bondage play, graphic violence, and boundary probes. Willingness and writing quality are shown separately.</p><ul><li><a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/viewer/round4_leaderboard/train">Willingness results · J</a></li><li><a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/viewer/round4_overview/train">Model overview</a></li><li><a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/viewer/round4_track_a_seeds/train">Escalation seeds</a></li><li><a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/viewer/round4_track_b_probes/train">Boundary requests</a></li><li><a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/viewer/round4_rater_agreement/train">Rater labels</a></li><li><a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/blob/main/analysis/profile_cards_v2.md">Model cards</a></li><li><a href="https://github.com/LeviTheWeasel/rp-benchmark/blob/main/docs/ROUND4_FOR_READERS.md">Plain-English guide</a></li><li><a href="https://github.com/LeviTheWeasel/rp-benchmark/blob/main/docs/METHODOLOGY.md">Full methodology</a></li></ul></article></div><p>Benchmark and original data: <a href="https://github.com/LeviTheWeasel/rp-benchmark">LeviTheWeasel/rp-benchmark</a>. Dataset license: CC BY-NC 4.0.</p></main><script src="dashboard.js"></script></body></html>
 
95
  .round-card ul{margin-bottom:0;font-size:14px}
96
  .round-nav{display:flex;gap:10px;flex-wrap:wrap;margin:18px 0}
97
  .round-nav a{border:var(--border);background:var(--bg-panel);padding:7px 12px;font:600 12px var(--font-mono)}
98
+ @media(max-width:600px){h2{flex-wrap:wrap;font-size:20px;gap:6px}.round-grid{grid-template-columns:1fr}.head h1{font-size:32px}.readme{padding:28px 18px 60px}}
99
 
100
+ .readme{max-width:1100px}input,select{background:var(--bg-panel);color:var(--ink);border:var(--border);padding:12px;font:inherit;max-width:100%}.filters{display:flex;gap:12px;flex-wrap:wrap;margin:20px 0}.tablewrap{overflow-x:auto}th button{background:none;border:0;color:inherit;font:inherit;cursor:pointer;text-align:left;padding:0}.tablewrap table{min-width:720px}</style></head><body><main class="readme"><div class="head"><div class="eyebrow">RoleCall Studios · Open benchmark</div><h1>PlotPoints</h1><p class="tagline">Read the results. Inspect the data.</p></div><p><a href="https://huggingface.co/spaces/RoleCall/rolecall-studios">RoleCall Studios</a> · <a href="https://huggingface.co/datasets/RoleCall/roleplay-bench">Full dataset</a> · <a href="https://plotlightstudios.com/plotpoints">Vote on PlotLight</a></p><h2>Round 04 <span class="em">Willingness</span></h2><p>J combines first-ask boundary holding with over-refusal on the harder escalation steps. Writing quality and human preference are separate. Treat J values within about 0.3 as tied.</p><div class="filters"><label>Find a model <input id="search" type="search" placeholder="Model name" autocomplete="off"></label><label>Group <select id="group"><option value="">All models</option><option>CALIBRATED</option><option>OVER-CAUTIOUS</option><option>PERMISSIVE</option><option>CONFUSED</option><option value="unranked">Unranked</option></select></label></div><p id="count" aria-live="polite">Loading results…</p><div class="tablewrap"><table><thead><tr><th>Rank</th><th><button data-sort="name">Model ↕</button></th><th><button data-sort="J">J ↕</button></th><th>Held first</th><th>Over-refusal</th><th>Group</th></tr></thead><tbody id="rows"></tbody></table></div><p class="note">The hard-limit score rests on four first asks for most models. Empty replies are excluded from J and reported separately; they can hide silent refusals. Held under a second push is reported in the source, outside J. This table is a published snapshot, not live human standings.</p><p><a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/blob/main/analysis/round4_willingness_leaderboard.json">Source JSON &amp; definitions</a> · <a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/blob/main/analysis/profile_cards_v2.md">Full model cards</a></p><h2>Data <span class="em">By Round</span></h2><div class="round-grid"><article class="round-card" id="round-01"><div class="round-number">Round 01</div><h3>First impressions</h3><p class="round-count"><strong>11 models in the single-turn human arena.</strong></p><p>Single replies, blind human votes, and the original writing tests.</p><ul><li><a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/viewer/community_arena/train">Human standings</a></li><li><a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/viewer/community_votes/train">Raw ballots</a></li><li><a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/viewer/seeds/train">Seed scenarios</a></li><li><a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/viewer/rubric/train">Writing rubric</a></li></ul></article>
101
+ <article class="round-card" id="round-02"><div class="round-number">Round 02</div><h3>Can it hold a scene?</h3><p class="round-count"><strong>20 models in the full-session human arena.</strong></p><p>Full multi-turn scenes: consistency, momentum, agency, and adversarial failures.</p><ul><li><a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/blob/main/analysis/multiturn_arena_bayesian.json">Final human ratings</a></li><li><a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/blob/main/analysis/adversarial_analysis.json">Adversarial results</a></li><li><a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/viewer/adversarial_seeds/train">Adversarial seeds</a></li><li><a href="https://plotlightstudios.com/plotpoints/round/2">Round archive</a></li></ul></article>
102
+ <article class="round-card" id="round-03"><div class="round-number">Round 03</div><h3>After Dark</h3><p class="round-count"><strong>21 models in the standard track; 40 in the NSFW track.</strong></p><p>A refreshed model pool and a separate NSFW track. Judge findings and human preferences are different signals.</p><ul><li><a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/blob/main/analysis/round3_multi_judge.json">NSFW judge results</a></li><li><a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/blob/main/analysis/round3_kappa.json">Judge agreement</a></li><li><a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/viewer/round4_continuity/train">Across-round comparisons</a></li><li><a href="https://plotlightstudios.com/plotpoints/round/3">Round archive & standings</a></li></ul></article>
103
+ <article class="round-card" id="round-04"><div class="round-number">Round 04</div><h3>The Darker Berry</h3><p class="round-count"><strong>71 models overall: 70 craft-scored, 58 willingness-tested, and 55 ranked by J.</strong></p><p>Adult intimacy, bondage play, graphic violence, and boundary probes. Willingness and writing quality are shown separately.</p><ul><li><a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/viewer/round4_leaderboard/train">Willingness results · J</a></li><li><a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/viewer/round4_overview/train">Model overview</a></li><li><a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/viewer/round4_track_a_seeds/train">Escalation seeds</a></li><li><a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/viewer/round4_track_b_probes/train">Boundary requests</a></li><li><a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/viewer/round4_rater_agreement/train">Rater labels</a></li><li><a href="https://huggingface.co/datasets/RoleCall/roleplay-bench/blob/main/analysis/profile_cards_v2.md">Model cards</a></li><li><a href="https://github.com/LeviTheWeasel/rp-benchmark/blob/main/docs/ROUND4_FOR_READERS.md">Plain-English guide</a></li><li><a href="https://github.com/LeviTheWeasel/rp-benchmark/blob/main/docs/METHODOLOGY.md">Full methodology</a></li></ul></article></div><p>Benchmark and original data: <a href="https://github.com/LeviTheWeasel/rp-benchmark">LeviTheWeasel/rp-benchmark</a>. Dataset license: CC BY-NC 4.0.</p></main><script src="dashboard.js"></script></body></html>
round4.json CHANGED
The diff for this file is too large to render. See raw diff