Spaces:
Running
Running
File size: 74,892 Bytes
a2f5229 c8d6a18 b7d9743 98b26a3 f472f2b 98b26a3 b7d9743 9ad4d81 b7d9743 c8d6a18 b7d9743 51e2750 b7d9743 9ad4d81 b7d9743 9ad4d81 b7d9743 9ad4d81 b7d9743 9ad4d81 51e2750 113b1d1 bcb4980 c11cd79 bcb4980 c11cd79 bcb4980 c11cd79 9ad4d81 c11cd79 b7d9743 9bda7f0 b7d9743 9bda7f0 9ad4d81 9bda7f0 9ad4d81 9bda7f0 9ad4d81 b7d9743 9bda7f0 b7d9743 9bda7f0 e206edc 9bda7f0 b7d9743 9bda7f0 b7d9743 9bda7f0 b7d9743 9ad4d81 9bda7f0 e206edc b7d9743 9bda7f0 b7d9743 9ad4d81 b7d9743 9bda7f0 9ad4d81 9bda7f0 b7d9743 9bda7f0 9ad4d81 9bda7f0 b7d9743 9bda7f0 b7d9743 9bda7f0 b7d9743 9bda7f0 b7d9743 9bda7f0 b7d9743 9bda7f0 b7d9743 9bda7f0 b7d9743 9bda7f0 b7d9743 9bda7f0 b7d9743 9bda7f0 b7d9743 9ad4d81 b7d9743 9bda7f0 9ad4d81 9bda7f0 b7d9743 9bda7f0 b7d9743 9bda7f0 9ad4d81 9bda7f0 b7d9743 9bda7f0 9ad4d81 9bda7f0 9ad4d81 9bda7f0 b7d9743 9bda7f0 9ad4d81 9bda7f0 9ad4d81 9bda7f0 9ad4d81 9bda7f0 b7d9743 9bda7f0 b7d9743 113b1d1 51e2750 113b1d1 51e2750 113b1d1 51e2750 113b1d1 51e2750 113b1d1 51e2750 113b1d1 51e2750 113b1d1 51e2750 113b1d1 bcb4980 c11cd79 bcb4980 c11cd79 bcb4980 c11cd79 bcb4980 c11cd79 bcb4980 c11cd79 bcb4980 c11cd79 9ad4d81 c11cd79 9ad4d81 c11cd79 9ad4d81 bcb4980 c11cd79 bcb4980 9ad4d81 b7d9743 9ad4d81 0995ae7 b7d9743 0995ae7 b7d9743 0995ae7 b7d9743 f472f2b b7d9743 c8d6a18 b7d9743 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 351 352 353 354 355 356 357 358 359 360 361 362 363 364 365 366 367 368 369 370 371 372 373 374 375 376 377 378 379 380 381 382 383 384 385 386 387 388 389 390 391 392 393 394 395 396 397 398 399 400 401 402 403 404 405 406 407 408 409 410 411 412 413 414 415 416 417 418 419 420 421 422 423 424 425 426 427 428 429 430 431 432 433 434 435 436 437 438 439 440 441 442 443 444 445 446 447 448 449 450 451 452 453 454 455 456 457 458 459 460 461 462 463 464 465 466 467 468 469 470 471 472 473 474 475 476 477 478 479 480 481 482 483 484 485 486 487 488 489 490 491 492 493 494 495 496 497 498 499 500 501 502 503 504 505 506 507 508 509 510 511 512 513 514 515 516 517 518 519 520 521 522 523 524 525 526 527 528 529 530 531 532 533 534 535 536 537 538 539 540 541 542 543 544 545 546 547 548 549 550 551 552 553 554 555 556 557 558 559 560 561 562 563 564 565 566 567 568 569 570 571 572 573 574 575 576 577 578 579 580 581 582 583 584 585 586 587 588 589 590 591 592 593 594 595 596 597 598 599 600 601 602 603 604 605 606 607 608 609 610 611 612 613 614 615 616 617 618 619 620 621 622 623 624 625 626 627 628 629 630 631 632 633 634 635 636 637 638 639 640 641 642 643 644 645 646 647 648 649 650 651 652 653 654 655 656 657 658 659 660 661 662 663 664 665 666 667 668 669 670 671 672 673 674 675 676 677 678 679 680 681 682 683 684 685 686 687 688 689 690 691 692 693 694 695 696 697 698 699 700 701 702 703 704 705 706 707 708 709 710 711 712 713 714 715 716 717 718 719 720 721 722 723 724 725 726 727 728 729 730 731 732 733 734 735 736 737 738 739 740 741 742 743 744 745 746 747 748 749 750 751 752 753 754 755 756 757 758 759 760 761 762 763 764 765 766 767 768 769 770 771 772 773 774 775 776 777 778 779 780 781 782 783 784 785 786 787 788 789 790 791 792 793 794 795 796 797 798 799 800 801 802 803 804 805 806 807 808 809 810 811 812 813 814 815 816 817 818 819 820 821 822 823 824 825 826 827 828 829 830 831 832 833 834 835 836 837 838 839 840 841 842 843 844 845 846 847 848 849 850 851 852 853 854 855 856 857 858 859 860 861 862 863 864 865 866 867 868 869 870 871 872 873 874 875 876 877 878 879 880 881 882 883 884 885 886 887 888 889 890 891 892 893 894 895 896 897 898 899 900 901 902 903 904 905 906 907 908 909 910 911 912 913 914 915 916 917 918 919 920 921 922 923 924 925 926 927 928 929 930 931 932 933 934 935 936 937 938 939 940 941 942 943 944 945 946 947 948 949 950 951 952 953 954 955 956 957 958 959 960 961 962 963 964 965 966 967 968 969 970 971 972 | <!doctype html>
<html lang="en">
<head>
<meta charset="utf-8" />
<meta name="viewport" content="width=device-width,initial-scale=1" />
<title>ORIS Research</title>
<style>
:root{
--bg:#08090b;--panel:#101216;--panel2:#15181d;--text:#f4f5f7;
--muted:#9fa5af;--line:rgba(255,255,255,.09);--accent:#e11d2e;
--soft:rgba(225,29,46,.11);--max:1160px;
}
*{box-sizing:border-box}
html{scroll-behavior:smooth}
body{margin:0;background:radial-gradient(circle at 85% -10%,rgba(225,29,46,.08),transparent 30%),var(--bg);color:var(--text);font-family:Inter,system-ui,-apple-system,BlinkMacSystemFont,"Segoe UI",sans-serif;line-height:1.65}
a{color:inherit;text-decoration:none}
.wrap{width:min(calc(100% - 36px),var(--max));margin:auto}
header{position:sticky;top:0;z-index:30;background:rgba(8,9,11,.9);backdrop-filter:blur(14px);border-bottom:1px solid var(--line)}
nav{min-height:66px;display:flex;align-items:center;justify-content:space-between;gap:20px}
.brand{font-weight:900;letter-spacing:.16em;font-size:14px}.brand span{color:var(--accent)}
.nav-note{font-size:12px;color:var(--muted)}
.hero{padding:92px 0 58px}
.eyebrow{color:var(--accent);font-size:11px;font-weight:800;letter-spacing:.18em;text-transform:uppercase}
h1{margin:10px 0 16px;font-size:clamp(58px,9vw,108px);line-height:.92;letter-spacing:-.055em;max-width:920px}
.hero p,.detail-hero p{max-width:840px;color:#c8cbd2;font-size:18px;margin:0}
.model-grid{display:grid;grid-template-columns:repeat(2,1fr);gap:16px;padding:22px 0 90px}
.model-card{position:relative;border:1px solid var(--line);background:var(--panel);padding:26px;min-height:245px;cursor:pointer;transition:.18s}
.model-card:hover{transform:translateY(-3px);border-color:rgba(225,29,46,.45);background:var(--panel2)}
.model-card::after{content:"↗";position:absolute;right:22px;top:18px;color:var(--muted)}
.model-card small{display:block;color:var(--accent);font-size:10px;text-transform:uppercase;letter-spacing:.12em;font-weight:800;margin-bottom:14px}
.model-card h2{margin:0 0 12px;font-size:34px;line-height:1;letter-spacing:-.035em}
.model-card p{margin:0;color:var(--muted);font-size:14px;max-width:520px}
.chips{display:flex;flex-wrap:wrap;gap:8px;margin-top:22px}
.chip{border:1px solid var(--line);background:#0c0e11;padding:6px 9px;font:11px ui-monospace,monospace;color:#c5c9d0}
.detail{display:none}.detail.active{display:block}
.detail-top{position:sticky;top:66px;z-index:20;background:rgba(8,9,11,.94);backdrop-filter:blur(14px);border-bottom:1px solid var(--line)}
.detail-bar{min-height:58px;display:flex;justify-content:space-between;align-items:center}
.back{border:1px solid var(--line);background:var(--panel);color:var(--text);padding:9px 13px;cursor:pointer}
.detail-name{font-size:12px;color:var(--muted)}
.detail-hero{padding:72px 0 50px}
.detail-hero h2{margin:8px 0 16px;font-size:clamp(48px,8vw,88px);line-height:.92;letter-spacing:-.05em}
section{padding:68px 0;border-top:1px solid var(--line)}
.section-title{margin:9px 0 24px;font-size:clamp(30px,4.5vw,46px);line-height:1.05;letter-spacing:-.04em;max-width:900px}
.prose{max-width:920px;color:#b8bcc4}.prose p{margin:0 0 18px}
.stats{display:grid;grid-template-columns:repeat(4,1fr);gap:12px;margin-top:28px}
.stat{border:1px solid var(--line);background:var(--panel);padding:20px}
.stat strong{display:block;font-size:28px;line-height:1.1;margin-bottom:5px}.stat span{color:var(--muted);font-size:12px}
.callout{margin-top:26px;border-left:3px solid var(--accent);background:var(--soft);padding:20px 22px;max-width:930px}
.callout p{margin:7px 0 0;color:#c3c7ce}
.timeline{margin-top:30px}
.timeline-item{display:grid;grid-template-columns:125px 1fr;gap:24px;padding-bottom:30px}
.timeline-year{color:var(--accent);font:800 12px ui-monospace,monospace}
.timeline-body{border-left:1px solid var(--line);padding-left:24px;position:relative}
.timeline-body:before{content:"";position:absolute;left:-5px;top:7px;width:9px;height:9px;border-radius:50%;background:var(--accent)}
.timeline-body h4{margin:0 0 6px;font-size:20px}.timeline-body p{margin:0;color:var(--muted)}
.table-wrap{overflow-x:auto;margin-top:26px}
table{width:100%;border-collapse:collapse;background:var(--panel);font-size:13px}
th,td{border:1px solid var(--line);padding:12px 13px;text-align:left}
th{color:var(--muted);font-size:10px;text-transform:uppercase;letter-spacing:.08em}
tr.highlight{background:var(--soft)}
.note{margin-top:9px;color:var(--muted);font-size:12px;max-width:920px}
.version-grid{display:grid;grid-template-columns:repeat(2,1fr);gap:14px;margin-top:28px}
.version-card{border:1px solid var(--line);background:var(--panel);padding:20px}
.version-card small{color:var(--accent);font-size:10px;text-transform:uppercase;letter-spacing:.1em}
.version-card h4{margin:8px 0;font-size:21px}.version-card p{margin:0;color:var(--muted);font-size:13px}
.update-log{display:grid;gap:10px;margin-top:30px;max-width:940px}
.update-item{display:grid;grid-template-columns:150px 1fr;gap:18px;border:1px solid var(--line);background:var(--panel);padding:15px 17px;cursor:pointer;transition:.18s}.update-item:hover{border-color:rgba(225,29,46,.5);background:var(--panel2);transform:translateY(-1px)}
.update-date{font:800 11px ui-monospace,monospace;color:var(--accent);letter-spacing:.03em;text-transform:uppercase}
.update-copy strong{display:block;font-size:14px;margin-bottom:2px}
.update-copy span{display:block;color:var(--muted);font-size:12px}
@media(max-width:900px){.update-item{grid-template-columns:1fr;gap:5px}}
footer{border-top:1px solid var(--line);padding:36px 0;color:var(--muted);font-size:12px}
@media(max-width:900px){.model-grid,.stats,.version-grid{grid-template-columns:1fr}.timeline-item{grid-template-columns:1fr;gap:8px}.nav-note{display:none}}
/* simplified research-index skin — closer to Horyzont */
:root{--bg:#fff;--panel:#fff;--panel2:#f8f8f8;--text:#111;--muted:#737373;--line:#e8e8e8;--accent:#111;--soft:#f6f6f6;--max:980px}
body{background:#fff;color:var(--text);font-family:Arial,Helvetica,sans-serif;line-height:1.5}
.wrap{width:min(calc(100% - 36px),var(--max))}
header{position:static;background:#fff;backdrop-filter:none;border-bottom:1px solid var(--line)}
nav{min-height:58px}.brand{letter-spacing:0;font-size:14px}.brand span{display:none}.nav-note{font-size:12px}
.hero{padding:78px 0 54px}.eyebrow{color:var(--muted);font-size:11px;letter-spacing:.08em;font-weight:400}
h1{font-size:clamp(52px,9vw,94px);line-height:.95;letter-spacing:-.06em;margin:0 0 18px;max-width:900px}
.hero p,.detail-hero p{font-size:17px;color:#333;max-width:760px}
.model-grid{display:block;padding:0 0 70px}.model-card{border:0;border-top:1px solid var(--line);background:#fff;padding:22px 0;min-height:0;cursor:pointer;display:grid;grid-template-columns:180px 1fr;column-gap:24px;align-items:start}
.model-card:last-child{border-bottom:1px solid var(--line)}.model-card:hover{transform:none;background:var(--soft);border-color:var(--line)}.model-card::after{right:10px;top:24px;color:#999}
.model-card small{grid-column:1;color:var(--muted);font-size:11px;letter-spacing:0;text-transform:none;margin:3px 0}.model-card h2{grid-column:2;margin:0 0 8px;font-size:25px;letter-spacing:-.03em}.model-card p{grid-column:2;color:#555;font-size:13px;max-width:650px}.model-card .chips{grid-column:2;margin-top:12px}
.chip{background:#fff;border:1px solid var(--line);padding:4px 7px;font-size:10px;color:#555}
.detail-top{position:static;background:#fff;backdrop-filter:none;border-bottom:1px solid var(--line)}.detail-bar{min-height:54px}.back{background:#fff;color:#111;border:0;padding:8px 0}.detail-hero{padding:62px 0 44px}.detail-hero h2{font-size:clamp(46px,8vw,78px);letter-spacing:-.055em}
section{padding:42px 0;border-top:1px solid var(--line)}.section-title{font-size:24px;letter-spacing:-.025em;margin:7px 0 18px}.prose{color:#444;max-width:820px}.prose p{margin-bottom:14px}
.stats{grid-template-columns:repeat(4,1fr);gap:8px}.stat,.version-card{background:#fff;border:1px solid var(--line);padding:16px}.stat strong{font-size:24px}.callout{background:var(--soft);border-left:2px solid #111;padding:16px 18px}.timeline-body:before{background:#111}.timeline-year{color:#555}.update-log{gap:0}.update-item{border:0;border-top:1px solid var(--line);background:#fff;padding:13px 0}.update-item:last-child{border-bottom:1px solid var(--line)}.update-item:hover{background:var(--soft);transform:none;border-color:var(--line)}.update-date{color:#777;letter-spacing:0;text-transform:none}
table{background:#fff}tr.highlight{background:var(--soft)}
details.tech{border-top:1px solid var(--line);padding:0}details.tech:last-child{border-bottom:1px solid var(--line)}details.tech summary{cursor:pointer;list-style:none;padding:17px 4px;font-size:14px;font-weight:700;display:flex;justify-content:space-between;gap:20px}details.tech summary::-webkit-details-marker{display:none}details.tech summary:after{content:"+";font-weight:400;color:#777}details.tech[open] summary:after{content:"−"}.tech-body{padding:0 4px 20px;max-width:850px;color:#444;font-size:13px}.tech-body p{margin:0 0 12px}.tech-grid{display:grid;grid-template-columns:180px 1fr;gap:0;border-top:1px solid var(--line);margin:14px 0}.tech-grid div{padding:10px 0;border-bottom:1px solid var(--line)}.tech-grid b{font-size:12px}.tech-grid span{font-size:12px;color:#555}pre.mini-code{overflow:auto;background:#f7f7f7;border:1px solid var(--line);padding:14px;font:12px/1.55 ui-monospace,SFMono-Regular,Menlo,Consolas,monospace;color:#222}code{font-family:ui-monospace,SFMono-Regular,Menlo,Consolas,monospace}.muted-line{color:#777;font-size:12px}
footer{padding:24px 0 40px}
@media(max-width:700px){.model-card{grid-template-columns:1fr;gap:4px}.model-card small,.model-card h2,.model-card p,.model-card .chips{grid-column:1}.stats{grid-template-columns:1fr 1fr}.tech-grid{grid-template-columns:1fr;gap:0}.tech-grid div:nth-child(odd){border-bottom:0;padding-bottom:2px}.tech-grid div:nth-child(even){padding-top:2px}}
</style>
</head>
<body>
<header><div class="wrap"><nav><a class="brand" href="#" onclick="showIndex();return false;">ORIS<span>.</span></a><div class="nav-note">Independent model research</div></nav></div></header>
<main id="index">
<div class="hero"><div class="wrap">
<div class="eyebrow">Model index</div>
<h1>Independent models. Shared research lineage.</h1>
<p>ORIS is a collection of independent experiments rather than one architecture. Choose a model to open its roadmap, versions, results and research notes.</p>
</div></div>
<div class="wrap model-grid">
<article class="model-card" onclick="showModel('polmath')">
<small>2024 · learning / custom-model research</small><h2>PolMATH</h2>
<p>The project where most of the early custom-model work happened: data preparation, tokenizer design, model internals, forward passes, numerical heads and difficult first attempts at exporting non-standard architectures to Hugging Face.</p>
<div class="chips"><span class="chip">custom architecture</span><span class="chip">numeric channel</span></div>
</article>
<article class="model-card" onclick="showModel('oris660')">
<small>2025 · training-model experiments</small><h2>ORIS 660M / Qwen 0.8B</h2>
<p>Two related training tracks: ORIS 660M kept the structural-recovery experiment; the Qwen 0.8B branch tested tokenizer replacement and training with part of the pretrained weights frozen.</p>
<div class="chips"><span class="chip">ORIS 660M</span><span class="chip">Qwen 0.8B</span><span class="chip">tokenizer swap</span><span class="chip">partial freezing</span></div>
</article>
<article class="model-card" onclick="showModel('smallc')">
<small>2026 · embedding / encoder research</small><h2>ORIS BERT</h2>
<p>Compact Polish BERT-style encoder developed for embeddings, fast local filtering, scoring and downstream fine-tuning.</p>
<div class="chips"><span class="chip">25.41M</span><span class="chip">local/global attention</span></div>
</article>
<article class="model-card" onclick="showModel('vyuhu')">
<small>2026 · active research</small><h2>Vyuhu</h2>
<p>One trained supernetwork, four deterministic compute profiles, and physically extractable models that reproduce their profile path exactly.</p>
<div class="chips">
<span class="chip">~280M first run planned</span>
<span class="chip">4 profiles</span>
<span class="chip">exact extraction</span>
<span class="chip">0.834B validation run passed</span>
</div>
</article>
<article class="model-card" onclick="showModel('vidar')">
<small>2026 · planned / active development</small><h2>Vidar-VL</h2>
<p>Planned ORIS vision-language model for understanding images and interacting with visual content in Polish.</p>
<div class="chips">
<span class="chip">image understanding</span>
<span class="chip">Polish vision-language</span>
<span class="chip">visual question answering</span>
<span class="chip">OCR / documents</span>
<span class="chip">multimodal reasoning</span>
</div>
</article>
<article class="model-card" onclick="showModel('orisvision')">
<small>2026 · embedding / vision encoder research</small><h2>ORIS Vision</h2>
<p>Visual counterpart to ORIS BERT: a compact embedding-oriented encoder for image similarity, retrieval, scoring and dataset filtering.</p>
<div class="chips">
<span class="chip">vision encoder</span>
<span class="chip">embeddings</span>
<span class="chip">image scoring</span>
<span class="chip">retrieval</span>
<span class="chip">visual filtering</span>
</div>
</article>
</div>
</main>
<article class="detail" id="model-polmath">
<div class="detail-top"><div class="wrap detail-bar"><button class="back" onclick="showIndex()">← All models</button><div class="detail-name">PolMATH</div></div></div>
<div class="wrap detail-hero">
<div class="eyebrow">Numerical representation research</div>
<h2>PolMATH</h2>
<p>A custom language-model experiment that became the main place to learn the complete model pipeline in practice — from data formatting and tokenizer training, through embeddings and transformer blocks, to forward passes, custom heads, decoding and the awkward reality of exporting non-standard solutions to Hugging Face. The numerical idea mattered, but so did learning how all the pieces actually fit together.</p>
<div class="update-log">
<div class="update-item"><div class="update-date">2024</div><div class="update-copy"><strong>Initial architecture and training experiments</strong><span>Custom numerical channel, numerical decoding, routing and number↔text association tests.</span></div></div>
<div class="update-item"><div class="update-date">Current</div><div class="update-copy"><strong>Publication-oriented rework</strong><span>Original branch paused; a cleaner protocol and English-language continuation are being prepared.</span></div></div>
</div>
</div>
<section><div class="wrap">
<div class="eyebrow">Why it existed</div>
<h3 class="section-title">Value and notation were treated as related, but not identical.</h3>
<div class="prose">
<p>PolMATH did not begin as an adapter attached to an existing pretrained model. The transformer stack, tokenizer behaviour, numerical codec, embedding fusion, auxiliary losses, routing logic and numerical decoding path were implemented as one experimental architecture.</p>
<p>Text could contain a dedicated numerical position while the actual value travelled through a structured numerical representation. That representation could encode integer and fractional digits, lengths, decimal form and exponent structure rather than collapsing every number into one scalar or leaving it entirely to BPE tokenization.</p>
<p>This made the experiment less about “doing mathematics” and more about asking whether a language model benefits from separating <em>what a number means</em> from <em>how that number happened to be written</em>.</p>
</div>
</div></section>
<section><div class="wrap">
<div class="eyebrow">Architecture</div>
<h3 class="section-title">Language decides that a number belongs here. A numerical path decides which number.</h3>
<div class="version-grid">
<div class="version-card"><small>Tokenizer</small><h4>[NUM] position</h4><p>Written numbers could be normalized into a dedicated numerical position while preserving linguistic context around them.</p></div>
<div class="version-card"><small>Numeric codec</small><h4>Structured representation</h4><p>Integer digits, fractional digits, notation flags and exponent structure were represented explicitly.</p></div>
<div class="version-card"><small>Fusion</small><h4>Gated numerical injection</h4><p>Numerical features were projected into the token representation and gated per position instead of affecting every token indiscriminately.</p></div>
<div class="version-card"><small>Objectives</small><h4>Multiple numerical heads</h4><p>Categorical structure prediction and a bucket/regression path provided complementary ways of reconstructing values.</p></div>
<div class="version-card"><small>Alignment</small><h4>Numeric-only pressure</h4><p>An auxiliary objective aligned text and numerical representations at number positions and penalized numerical leakage outside them.</p></div>
<div class="version-card"><small>Generation</small><h4>Two-stage output</h4><p>The LM could first decide that a numerical token belongs in the sequence, then the numerical head could reconstruct the value.</p></div>
</div>
</div></section>
<section><div class="wrap">
<div class="eyebrow">Technical notebook</div>
<h3 class="section-title">The useful part is in the plumbing.</h3>
<div class="prose"><p>PolMATH was the project where the author spent the most time taking the pipeline apart and putting it back together. The point of the notes below is not to dump the repository onto a webpage, but to expose the decisions that normally disappear behind <code>AutoModel.from_pretrained()</code>.</p></div>
<details class="tech" open><summary>Tokenizer · ByteLevel-BPE + a real numerical position</summary><div class="tech-body">
<p>The tokenizer branch trains a ByteLevel-BPE vocabulary while keeping <code>[PAD]</code>, <code>[BOS]</code>, <code>[EOS]</code>, <code>[NUM]</code> and the ten digits available explicitly. A regex detects ordinary decimals, decimal commas, scientific notation and suffixes such as <code>1930s</code> or <code>10th</code>. The numeric magnitude is moved out of the text stream and the text receives <code>[NUM]</code> in its place.</p>
<pre class="mini-code">pre, spans = self._preprocess_text_for_numbers(raw_text)
enc = self._tok.encode(pre)
# token stream: "Temperature is [NUM] degrees"
# side channel: {position_of_NUM: "23.5"}</pre>
<p class="muted-line">This makes the tokenizer responsible for linguistic placement, while the numeric path can carry the value itself.</p>
</div></details>
<details class="tech"><summary>Number codec · structure instead of one floating-point scalar</summary><div class="tech-body">
<p>The categorical codec decomposes a number into notation flags, integer/fraction/exponent lengths, exponent sign and digit classes. In the current magnitude-only branch the leading value sign is intentionally separated from the numeric magnitude.</p>
<div class="tech-grid"><div><b>Integer digits</b></div><div><span>up to 10 explicit digit slots</span></div><div><b>Fraction</b></div><div><span>up to 6 digit slots + decimal comma flag</span></div><div><b>Exponent</b></div><div><span>up to 3 digits + exponent sign</span></div><div><b>Notation flags</b></div><div><span>scientific / fraction / decimal form</span></div></div>
<pre class="mini-code">1939.0 → int=[1,9,3,9], frac=[0]
1.939e3 → int=[1], frac=[9,3,9], exp=[3]
# different notation, recoverable structure</pre>
</div></details>
<details class="tech"><summary>Embedding fusion · let the model decide how much number to inject</summary><div class="tech-body">
<p>The normal token, position and token-type embeddings are built first. Numeric features are projected to hidden size. In the gated variant a small MLP sees both representations and produces a per-position gate; the numeric mask suppresses the channel everywhere except numerical positions.</p>
<pre class="mini-code">proj = self.numeric_proj(numeric_features)
g = sigmoid(self.gate_mlp(cat([x_base, proj])))
g = g * numeric_mask.unsqueeze(-1)
x = x_base + g * proj</pre>
<p>An auxiliary term experiments with aligning text and numeric representations where a number exists and penalizing numeric leakage elsewhere.</p>
</div></details>
<details class="tech"><summary>Transformer body and forward · deliberately ordinary where it should be ordinary</summary><div class="tech-body">
<p>The language backbone is intentionally understandable: causal scaled dot-product self-attention, a triangular mask, pre-norm residual blocks and a GELU MLP. One default configuration in the branch uses a 32k vocabulary, 768 hidden width, 12 layers, 12 attention heads, 3072 intermediate width and 2048 positions.</p>
<pre class="mini-code">h = self.ln1(x)
x = x + self.attn(h, attn_mask=attention_mask)
h2 = self.ln2(x)
x = x + self.mlp(h2)</pre>
<p>This was useful pedagogically: custom behaviour was isolated around numbers rather than hiding every component behind another abstraction.</p>
</div></details>
<details class="tech"><summary>Heads + router · categorical reconstruction versus bucket/regression</summary><div class="tech-body">
<p>The model can emit ordinary vocabulary logits and numerical predictions from the same hidden states. The categorical head predicts the components needed to reconstruct notation. An optional mixed head adds bucket classification and regression. A learned two-way router tests whether the model can choose which numerical path should dominate instead of hard-coding the choice.</p>
<pre class="mini-code">logits = self.lm_head(x)
pred_cat = self.numeric_head_cat(x)
pred_mix = self.numeric_head_mix(x)
gate = softmax(self.router(x), dim=-1)</pre>
</div></details>
<details class="tech"><summary>Tests · what gets checked before a long run</summary><div class="tech-body">
<p>The tokenizer script includes a direct sanity check on special-token IDs, digit IDs, encoded tokens, numeric positions and reconstruction. One test string deliberately mixes arithmetic, a negative decimal, a year-like form and an ordinal:</p>
<pre class="mini-code">"8+9 is 17. Temperature is -23.5 degrees.
In the 1930s, 10th place was okay."</pre>
<p>Useful test families for this branch include round-trip encode/decode, notation-equivalence pairs, sign and suffix edge cases, scientific notation, numeric-mask leakage, gate behaviour, shape checks for every head, causal-mask checks, and reversed association prompts such as fact → year versus year → fact.</p>
</div></details>
<details class="tech"><summary>Hugging Face exports · where custom ideas stop being cute</summary><div class="tech-body">
<p>A large part of the learning curve was not the forward pass itself but packaging a custom configuration, tokenizer behaviour, extra numeric tensors and custom model outputs so that they behaved predictably outside the local training script. Early exports were inconsistent: a model can train locally and still fail the practical <em>save → load → tokenize → forward → generate</em> path expected by Hugging Face tooling.</p>
<p>The lesson from PolMATH was blunt: a custom architecture is not finished when <code>loss.backward()</code> works. Serialization, configuration fields, tokenizer special tokens, output schemas and reload tests are part of the architecture too.</p>
</div></details>
</div></section>
<section><div class="wrap">
<div class="eyebrow">Observed training behaviour</div>
<h3 class="section-title">The numerical channel became usable rather than decorative.</h3>
<div class="prose">
<p>The training pipeline reached the intended qualitative behaviours: matching number words to numeric values and back, deciding when a numerical position should appear despite ordinary text tokens dominating the corpus, and learning useful associations between values and facts.</p>
<p>One of the useful examples was the relation between <strong>1939</strong> and the outbreak of the Second World War. The interesting part was not only producing the number after the fact, but also retaining the association when the direction of the prompt was reversed.</p>
<p>Different written forms such as <strong>1939</strong>, <strong>1939.0</strong> and <strong>1.939e3</strong> could be learned as related expressions of the same value rather than as unrelated text fragments.</p>
</div>
<div class="callout"><strong>Important limitation</strong><p>PolMATH was not a general-purpose mathematical reasoner. Its value was narrower: it demonstrated that an explicit numerical path could coexist with ordinary language modelling and learn useful number-related behaviour.</p></div>
</div></section>
<section><div class="wrap">
<div class="eyebrow">Roadmap</div>
<h3 class="section-title">Paused, reorganized, and no longer intended to remain Polish-only.</h3>
<div class="timeline">
<div class="timeline-item"><div class="timeline-year">2024</div><div class="timeline-body"><h4>Initial PolMATH</h4><p>Custom architecture and structured numerical representation experiments.</p></div></div>
<div class="timeline-item"><div class="timeline-year">2024–25</div><div class="timeline-body"><h4>Training and data experiments</h4><p>Numeric placement, notation equivalence, routing, generation and association tests.</p></div></div>
<div class="timeline-item"><div class="timeline-year">Current</div><div class="timeline-body"><h4>Original branch paused</h4><p>The work is being cleaned up rather than simply continued as another checkpoint.</p></div></div>
<div class="timeline-item"><div class="timeline-year">Next</div><div class="timeline-body"><h4>Publication-oriented continuation</h4><p>A cleaner experimental protocol and an English-language iteration are planned so the idea can be evaluated beyond a Polish-only setting.</p></div></div>
</div>
</div></section>
</article>
<article class="detail" id="model-oris660">
<div class="detail-top"><div class="wrap detail-bar"><button class="back" onclick="showIndex()">← All models</button><div class="detail-name">ORIS 660M / Qwen 0.8B</div></div></div>
<div class="wrap detail-hero">
<div class="eyebrow">Structural compression and recovery</div>
<h2>ORIS 660M / Qwen 0.8B</h2>
<p>A pair of practical model-training experiments. ORIS 660M is the structural-compression and recovery branch described below; the earlier Qwen 0.8B track focused on changing the tokenizer and continuing training while freezing part of the pretrained model.</p>
<div class="stats"><div class="stat"><strong>660.13M</strong><span>parameters</span></div><div class="stat"><strong>32 → 12</strong><span>transformer blocks</span></div><div class="stat"><strong>621.984M</strong><span>benchmarked recovery labels</span></div><div class="stat"><strong>Paused</strong><span>research active</span></div></div>
<div class="update-log">
<div class="update-item"><div class="update-date">07 Aug 2026</div><div class="update-copy"><strong>ORIS 660M begins</strong><span>Depth-reduction experiment from Bielik-1.5B-v3.</span></div></div>
<div class="update-item"><div class="update-date">Recovery</div><div class="update-copy"><strong>Selected-vs-uniform architecture tests</strong><span>Layer identity proved important to recovery trajectory.</span></div></div>
<div class="update-item"><div class="update-date">Later work</div><div class="update-copy"><strong>Knowledge and data-control experiments</strong><span>Inherited-vs-random controls, Polish diagnostics and a staged 22B-token continuation plan.</span></div></div>
</div>
</div>
<section><div class="wrap">
<div class="eyebrow">2025 · Qwen 0.8B track</div>
<h3 class="section-title">Change the tokenizer, freeze part of the model, then see what actually adapts.</h3>
<div class="prose">
<p>The Qwen 0.8B experiment was a more conventional pretrained-model surgery than ORIS 660M. The working test replaced the tokenizer and continued training while keeping part of the inherited weights frozen. The practical question was how much of a pretrained model can be preserved when the interface between raw text and embeddings changes.</p>
<p>This branch was useful less as a final model and more as training practice: vocabulary replacement, embedding adaptation, deciding which modules should remain frozen, checkpoint compatibility and observing where transfer breaks when the tokenizer no longer matches the one used during pretraining.</p>
</div>
<div class="callout"><strong>Scope of the note</strong><p>The uploaded material documents the ORIS 660M branch in detail; the tokenizer-replacement / partial-freezing Qwen description here follows the project history supplied by the author, not a benchmark table from the provided files.</p></div>
</div></section>
<section><div class="wrap">
<div class="eyebrow">Origin</div>
<h3 class="section-title">Not a conventionally scaled-down 660M transformer.</h3>
<div class="prose">
<p>The lineage is <strong>Qwen 2.5 → Bielik-1.5B-v3 → ORIS 660M</strong>. The starting teacher had 32 transformer blocks. ORIS retained only 12 while preserving the teacher's hidden width, attention geometry, tokenizer, embeddings, feed-forward dimensions and LM head.</p>
<p>The selected blocks were <strong>0, 1, 2, 3, 4, 21, 23, 24, 25, 29, 30 and 31</strong>. They were not chosen by simply keeping every second or third layer. Small ablations and calibration tests were used to identify less-sensitive regions and compare candidate 9–12 layer structures.</p>
<p>The extreme jump from teacher layer 4 to layer 21 means that downstream blocks receive representations unlike those they originally saw during pretraining. That mismatch is one of the central features of the experiment.</p>
</div>
</div></section>
<section><div class="wrap">
<div class="eyebrow">Step 0</div>
<h3 class="section-title">The student became faster and lighter — and language modelling collapsed.</h3>
<div class="table-wrap"><table>
<thead><tr><th>Model</th><th>Parameters</th><th>Loss</th><th>Perplexity</th><th>Throughput</th><th>Peak VRAM</th></tr></thead>
<tbody>
<tr class="highlight"><td>ORIS 660M step0</td><td>660M</td><td>7.1613</td><td>1288.58</td><td>17.29k tok/s</td><td>1.54 GiB</td></tr>
<tr><td>Bielik-1.5B</td><td>1.596B</td><td>2.2692</td><td>9.67</td><td>6.69k tok/s</td><td>3.29 GiB</td></tr>
</tbody>
</table></div>
<p class="note">Fixed evaluation on 128 sequences of length 1024.</p>
<div class="callout"><strong>The point was not pruning without damage.</strong><p>The network was strongly damaged. The experiment became: can an inherited but structurally broken network reorganize itself through ordinary next-token training?</p></div>
</div></section>
<section><div class="wrap">
<div class="eyebrow">Architecture selection</div>
<h3 class="section-title">The identity of retained layers materially changed recovery.</h3>
<div class="table-wrap"><table>
<thead><tr><th>Architecture</th><th>Initial validation loss</th><th>After ~2.03M training tokens</th></tr></thead>
<tbody>
<tr class="highlight"><td>Selected 12-layer ORIS</td><td>7.2737</td><td>4.3134</td></tr>
<tr><td>Uniform 12-layer control</td><td>8.9560</td><td>5.4880</td></tr>
<tr><td>Selected 10-layer candidate</td><td>—</td><td>~4.9100</td></tr>
</tbody>
</table></div>
<div class="prose" style="margin-top:24px">
<p>These experiments were intentionally small and do not establish that the chosen structure is globally optimal. They do show that equal depth does not imply equal recoverability.</p>
<p>Main recovery used standard causal language modelling. Teacher/student KL was useful for screening candidate structures, but teacher logits were not the training objective during the main continuation run.</p>
</div>
<div class="stats"><div class="stat"><strong>589.248M</strong><span>earlier recovery labels</span></div><div class="stat"><strong>32.736M</strong><span>V2.5A continuation labels</span></div><div class="stat"><strong>621.984M</strong><span>cumulative benchmarked labels</span></div><div class="stat"><strong>~1×</strong><span>roughly one recovery label per student parameter</span></div></div>
</div></section>
<section><div class="wrap">
<div class="eyebrow">Generation behaviour</div>
<h3 class="section-title">Loss recovered faster than stable generation.</h3>
<div class="prose">
<p>Continued training restored grammatical Polish, reasonable local continuations and recognizable semantic associations, but free generation remained unstable.</p>
<p>Observed issues included greedy repetition loops, topic drift, weak long-range coherence, hallucinated dates and numerical sequences, web/forum residue, metadata-like fragments, premature EOS at some checkpoints and high sensitivity to sampling strategy.</p>
<p>One of the strongest lessons was that <strong>lower loss did not translate monotonically into better generation</strong>. Syntax, factual accessibility, calibration, semantic control and free-generation stability could improve at different rates.</p>
</div>
</div></section>
<section><div class="wrap">
<div class="eyebrow">V2.5A benchmark snapshot</div>
<h3 class="section-title">A recovery snapshot, not a leaderboard claim.</h3>
<div class="table-wrap"><table>
<thead><tr><th>Task</th><th>ORIS V2.5A</th></tr></thead>
<tbody>
<tr><td>Belebele accuracy</td><td>0.2256</td></tr>
<tr><td>8tags accuracy</td><td>0.1757</td></tr>
<tr><td>PoLeMo2 in accuracy</td><td>0.4155</td></tr>
<tr><td>PoLeMo2 out accuracy</td><td>0.3684</td></tr>
<tr><td>DYK binary F1</td><td>0.1268</td></tr>
<tr><td>PSC binary F1</td><td>0.1627</td></tr>
<tr><td>PPC accuracy</td><td>0.4180</td></tr>
<tr><td>CBD macro F1</td><td>0.1291</td></tr>
<tr><td>KLEJ NER accuracy</td><td>0.1025</td></tr>
<tr class="highlight"><td>PolQA reranking</td><td>0.5157</td></tr>
</tbody>
</table></div>
</div></section>
<section><div class="wrap">
<div class="eyebrow">Later diagnostics</div>
<h3 class="section-title">Competitive signals and very obvious failure modes appeared together.</h3>
<h4>Project-specific MC / continuation probe</h4>
<div class="table-wrap"><table>
<thead><tr><th>Model</th><th>Params</th><th>Raw MC</th><th>Normalized MC</th><th>Continuation NLL</th></tr></thead>
<tbody><tr class="highlight"><td>ORIS 660M</td><td>660M</td><td>0.750</td><td>0.500</td><td>3.3525</td></tr><tr><td>Qra-1b</td><td>~1.10B</td><td>0.917</td><td>0.583</td><td>1.5715</td></tr></tbody>
</table></div>
<p class="note">Qra remained clearly stronger; the comparison tested whether ORIS had returned to a meaningful operating regime.</p>
<h4 style="margin-top:34px">SpeakLeash polish_mc diagnostic — 0-shot, 100 examples per task</h4>
<div class="table-wrap"><table>
<thead><tr><th>Model</th><th>Parameters</th><th>Accuracy</th><th>Normalized accuracy</th><th>F1</th></tr></thead>
<tbody><tr class="highlight"><td>ORIS 660M</td><td>660M</td><td>0.399</td><td>0.386</td><td>0.0116</td></tr><tr><td>APT3-1B-Base</td><td>~1B</td><td>0.330</td><td>0.300</td><td>0.3215</td></tr></tbody>
</table></div>
<div class="table-wrap"><table>
<thead><tr><th>Task</th><th>ORIS</th><th>APT3</th></tr></thead>
<tbody>
<tr><td>PoLeMo2 in accuracy</td><td><strong>0.43</strong></td><td>0.34</td></tr>
<tr><td>PoLeMo2 out accuracy</td><td>0.33</td><td><strong>0.38</strong></td></tr>
<tr><td>8tags accuracy</td><td>0.12</td><td><strong>0.20</strong></td></tr>
<tr><td>Belebele accuracy</td><td>0.23</td><td><strong>0.25</strong></td></tr>
<tr><td>KLEJ NER accuracy</td><td><strong>0.31</strong></td><td>0.05</td></tr>
<tr><td>PolQA reranking accuracy</td><td><strong>0.62</strong></td><td>0.39</td></tr>
<tr><td>PPC accuracy</td><td><strong>0.44</strong></td><td>0.42</td></tr>
<tr><td>PSC accuracy</td><td><strong>0.68</strong></td><td>0.32</td></tr>
</tbody>
</table></div>
<div class="callout"><strong>The shape of the errors mattered more than the aggregate.</strong><p>ORIS could show surprisingly high accuracy on some tasks while binary F1 collapsed because of severe class bias. That is exactly why these results are diagnostic rather than a claim of general superiority over APT3.</p></div>
</div></section>
<section><div class="wrap">
<div class="eyebrow">Knowledge recovery</div>
<h3 class="section-title">Was knowledge destroyed, degraded, inaccessible — or relearned?</h3>
<div class="prose">
<p>An inherited 12-layer ORIS student was compared against an architecture-matched model initialized from scratch. Both had comparable structural capacity; only one inherited the teacher's pretrained parameter structure.</p>
</div>
<div class="stats"><div class="stat"><strong>24.7 → 35.0%</strong><span>inherited fact-level majority accuracy</span></div><div class="stat"><strong>0.154 → ~0.366</strong><span>teacher-ranking Spearman</span></div><div class="stat"><strong>~10%</strong><span>random control accuracy</span></div><div class="stat"><strong>-0.109</strong><span>random control Spearman</span></div></div>
<div class="table-wrap"><table><thead><tr><th>ENTITY_ONLY subset</th><th>Before</th><th>After</th></tr></thead><tbody><tr class="highlight"><td>Accuracy</td><td>20.0%</td><td>41.5%</td></tr><tr><td>Mean factual margin</td><td>-0.840</td><td>+0.120</td></tr></tbody></table></div>
<div class="prose" style="margin-top:24px">
<p>The result does not prove that complete symbolic facts remain intact inside individual weights. It does suggest that the inherited network begins recovery from a qualitatively different state than a random model with the same architecture.</p>
<p>A small generation probe captured the ambiguity: “Polska jest…” remained approximately correct, while “Kopernik był…” produced <strong>“Kopernik był w kosmosie.”</strong> The statement is false, but the astronomy/space semantic neighborhood survived.</p>
</div>
<div class="version-grid">
<div class="version-card"><small>Case 1</small><h4>Accessible</h4><p>Knowledge survives and remains directly usable.</p></div>
<div class="version-card"><small>Case 2</small><h4>Degraded</h4><p>The semantic region survives while precise retrieval fails.</p></div>
<div class="version-card"><small>Case 3</small><h4>Latent</h4><p>Useful structure may survive but no longer be reachable through the altered path.</p></div>
<div class="version-card"><small>Case 4</small><h4>Relearned</h4><p>The information genuinely has to be reacquired during continued training.</p></div>
</div>
</div></section>
<section><div class="wrap">
<div class="eyebrow">Data research</div>
<h3 class="section-title">The bottleneck moved from model architecture to controlling what the recovering model sees.</h3>
<div class="prose">
<p>Later generations repeatedly exposed artifacts from web-derived training data: forum fragments, metadata, SEO text, navigation elements, date-heavy sequences, list structures, transcripts, document templates and poorly separated topic boundaries.</p>
<p>The newer pipeline therefore separates linguistic quality, structural quality, corruption, noise, difficulty, domain, information density and <strong>knowledge value</strong> rather than compressing all of them into one score.</p>
<p>A difficult scientific or legal document can have high perplexity and many rare tokens while still carrying high knowledge value. A fluent generic paragraph can look clean while contributing almost none.</p>
</div>
<div class="version-grid">
<div class="version-card"><small>01</small><h4>Canonical ingest</h4><p>Stable IDs, provenance, validation, sharding and raw preservation.</p></div>
<div class="version-card"><small>02</small><h4>Repair & structure</h4><p>Encoding repair, paragraph preservation, segmentation and conservative local deduplication.</p></div>
<div class="version-card"><small>03</small><h4>Signal extraction</h4><p>Language confidence, coherence, repetition, information density, rare-token statistics and perplexity.</p></div>
<div class="version-card"><small>04</small><h4>Multi-axis classification</h4><p>Separate estimates for quality, noise, knowledge value and reconstruction decisions.</p></div>
<div class="version-card"><small>05</small><h4>Knowledge structure</h4><p>Domain hierarchies, subdomains and knowledge flags.</p></div>
<div class="version-card"><small>06</small><h4>Corpus curation</h4><p>Global exact/fuzzy deduplication, redundancy control and later curriculum balancing.</p></div>
</div>
</div></section>
<section><div class="wrap">
<div class="eyebrow">Snowball hypothesis</div>
<h3 class="section-title">Repeated small corpus defects may become directional model errors.</h3>
<div class="prose">
<p>The working hypothesis is not that one bad example damages a foundation model. The concern is cumulative: an unstable model repeatedly sees similar defects, small representation errors reinforce each other, and later data is interpreted through an already-shifted internal state.</p>
<p>In that view, <strong>data ordering can matter as much as data inclusion</strong>.</p>
</div>
</div></section>
<section><div class="wrap">
<div class="eyebrow">Planned V3 continuation</div>
<h3 class="section-title">22B tokens staged by what the recovering network needs next.</h3>
<div class="timeline">
<div class="timeline-item"><div class="timeline-year">S1 · 6B</div><div class="timeline-body"><h4>Stabilization and structural recovery</h4><p>Coherent, lower-risk data intended to restore reliable autoregressive behaviour.</p></div></div>
<div class="timeline-item"><div class="timeline-year">S2 · 12B</div><div class="timeline-body"><h4>Main knowledge recovery and expansion</h4><p>Broader domains, higher information density and more difficult material after stabilization.</p></div></div>
<div class="timeline-item"><div class="timeline-year">S3 · 4B</div><div class="timeline-body"><h4>Consolidation and robustness</h4><p>Harder distributions introduced under tighter control.</p></div></div>
</div>
<p class="note">This curriculum is proposed, not completed.</p>
</div></section>
<section><div class="wrap">
<div class="eyebrow">Why the branch is paused</div>
<h3 class="section-title">Because simply feeding the checkpoint more web text would answer the least interesting questions.</h3>
<div class="prose">
<p>ORIS is an independent, self-funded project developed primarily on privately owned compute. A substantial portion of pruning, validation, training and evaluation was performed locally, including on an RTX 5060 Ti.</p>
<p>The existing experiments suggest that a severely depth-reduced Polish model can recover useful language-model behaviour. The harder questions are now what exactly survives structural reduction, what is genuinely relearned, whether inherited knowledge can be distinguished from reacquired knowledge, and whether recovery can be controlled through data ordering at 10B–100B+ token scale.</p>
</div>
<div class="callout"><strong>Paused does not mean failed.</strong><p>The experiment clarified the next bottleneck. Work shifted toward data infrastructure, evaluation methodology, knowledge-recovery controls and additional compute before committing the 660M branch to a much larger run.</p></div>
</div></section>
</article>
<article class="detail" id="model-smallc">
<div class="detail-top"><div class="wrap detail-bar"><button class="back" onclick="showIndex()">← All models</button><div class="detail-name">ORIS BERT</div></div></div>
<div class="wrap detail-hero">
<div class="eyebrow">Compact Polish encoder</div>
<h2>ORIS BERT</h2>
<p>ORIS BERT is the continuation of the model originally developed under the working name ORIS Bert Small C. It uses BERT-style masked-language pretraining and is treated primarily as a compact encoder / embedding model, not a stock scaled-down BERT.</p>
<div class="stats"><div class="stat"><strong>25.41M</strong><span>parameters</span></div><div class="stat"><strong>6</strong><span>layers</span></div><div class="stat"><strong>128K</strong><span>Polish-oriented BPE</span></div><div class="stat"><strong>8.00B</strong><span>input pretraining tokens</span></div></div>
<div class="update-log">
<div class="update-item"><div class="update-date">16 Aug 2026</div><div class="update-copy"><strong>Small C checkpoint completed</strong><span>From-scratch 8B-token MLM pretraining completed.</span></div></div>
<div class="update-item"><div class="update-date">Downstream</div><div class="update-copy"><strong>KLEJ-style and document-filtering tests</strong><span>Established both the model's limitations and its practical pipeline advantage.</span></div></div>
<div class="update-item"><div class="update-date">Efficiency</div><div class="update-copy"><strong>Local encoder benchmarking</strong><span>Measured strong throughput and VRAM advantages on RTX 5060 Ti / Blackwell-oriented workloads.</span></div></div>
</div>
</div>
<section><div class="wrap">
<div class="eyebrow">Why this model exists</div>
<h3 class="section-title">A production bottleneck accidentally became an architecture project.</h3>
<div class="prose">
<p>Dataset preparation for ORIS 660M became a throughput bottleneck, so a smaller encoder was built specifically for local filtering, categorization and scoring on NVIDIA Blackwell hardware.</p>
<p>What began as an internal pipeline tool became a compact Polish encoder intended primarily as a backbone for task-specific fine-tuning.</p>
</div>
<div class="callout"><strong>Not a sentence-embedding model out of the box.</strong><p>Raw mean-pooled embeddings are strongly anisotropic and are not recommended for zero-shot semantic search without additional contrastive or task-specific fine-tuning.</p></div>
</div></section>
<section><div class="wrap">
<div class="eyebrow">Architecture</div>
<h3 class="section-title">The “BERT” part describes the training style more than the architecture.</h3>
<div class="table-wrap"><table>
<thead><tr><th>Property</th><th>Value</th></tr></thead>
<tbody>
<tr><td>Model type</td><td>Custom Transformer encoder</td></tr>
<tr><td>Hidden size</td><td>384</td></tr>
<tr><td>Token embedding size</td><td>128</td></tr>
<tr><td>Attention heads</td><td>6</td></tr>
<tr><td>Context length</td><td>1024</td></tr>
<tr class="highlight"><td>Attention layout</td><td>256, 256, 1024, 256, 256, 256</td></tr>
<tr><td>Normalization</td><td>RMSNorm</td></tr>
<tr><td>Objective</td><td>Masked Language Modeling</td></tr>
<tr><td>Initialization</td><td>From random initialization</td></tr>
</tbody>
</table></div>
<div class="prose" style="margin-top:24px">
<p>Five of six layers use local 256-token attention. The third layer performs full 1024-token attention and acts as the main global mixing layer.</p>
<p>The vocabulary is intentionally large, but the token embedding dimension is only 128 while the encoder hidden size is 384. This factorization keeps the 128K vocabulary from dominating the parameter budget.</p>
<p>During MLM pretraining, vocabulary logits are calculated only for masked positions. Those hidden states are projected from 384 dimensions back into the 128-dimensional token space and scored using the tied token-embedding matrix.</p>
</div>
</div></section>
<section><div class="wrap">
<div class="eyebrow">Pretraining</div>
<h3 class="section-title">8.00B input tokens from scratch on a single local GPU.</h3>
<div class="table-wrap"><table>
<thead><tr><th>Setting</th><th>Value</th></tr></thead>
<tbody>
<tr><td>Sequence length</td><td>1024</td></tr>
<tr><td>Micro-batch size</td><td>16</td></tr>
<tr><td>Gradient accumulation</td><td>8</td></tr>
<tr><td>Tokens / optimizer update</td><td>131,072</td></tr>
<tr><td>Optimizer updates</td><td>61,036</td></tr>
<tr><td>Peak learning rate</td><td>3e-4</td></tr>
<tr><td>Mask probability</td><td>15%</td></tr>
<tr><td>Precision</td><td>BF16 autocast</td></tr>
<tr><td>GPU</td><td>RTX 5060 Ti 16GB</td></tr>
<tr><td>Training time</td><td>~13.44 h</td></tr>
<tr class="highlight"><td>Average throughput</td><td>~165.4K input tok/s</td></tr>
</tbody>
</table></div>
<div class="prose" style="margin-top:24px">
<p>Training data was primarily Polish MADLAD with an auxiliary mixture containing Wikipedia, OpenSubtitles PL, balanced NKJP and Polish legal/judicial text. The auxiliary pool represented roughly 15% of generated training sequences.</p>
<p>The tokenizer is a custom 128K BPE with NFKC normalization and Metaspace pre-tokenization. It is Polish-oriented but includes a broad Unicode alphabet for noisy web text.</p>
</div>
</div></section>
<section><div class="wrap">
<div class="eyebrow">Sentence-space limitation</div>
<h3 class="section-title">The raw encoder space is highly anisotropic under mean pooling.</h3>
<div class="table-wrap"><table>
<thead><tr><th>Model</th><th>Mean cosine for unrelated texts</th></tr></thead>
<tbody><tr class="highlight"><td>ORIS BERT</td><td>~0.987</td></tr><tr><td>HerBERT</td><td>~0.90</td></tr><tr><td>PolDense</td><td>~0.26</td></tr></tbody>
</table></div>
<p class="note">Internal diagnostic only; not a general encoder-quality benchmark.</p>
</div></section>
<section><div class="wrap">
<div class="eyebrow">Polish downstream benchmarks</div>
<h3 class="section-title">Competitive on several tasks, clearly weaker on others — at about one quarter of PolBERTa's size.</h3>
<div class="table-wrap"><table>
<thead><tr><th>Task</th><th>Metric</th><th>ORIS BERT</th><th>PolBERTa base</th></tr></thead>
<tbody>
<tr><td>NKJP-NER</td><td>Macro-F1</td><td>75.52</td><td><strong>84.36</strong></td></tr>
<tr class="highlight"><td>CDSC-E</td><td>Accuracy</td><td><strong>91.30</strong></td><td>91.00</td></tr>
<tr><td>CDSC-R</td><td>Spearman</td><td>88.18</td><td><strong>88.97</strong></td></tr>
<tr class="highlight"><td>CBD</td><td>F1(+)</td><td><strong>50.24</strong></td><td>43.75</td></tr>
<tr><td>PolEmo2.0-IN</td><td>Accuracy</td><td>83.33</td><td><strong>85.32</strong></td></tr>
<tr class="highlight"><td>PolEmo2.0-OUT</td><td>Accuracy</td><td><strong>65.59</strong></td><td>63.77</td></tr>
<tr><td>DYK</td><td>F1(+)</td><td>37.86</td><td><strong>46.31</strong></td></tr>
<tr><td>PSC</td><td>Macro-F1</td><td>57.28</td><td><strong>85.87</strong></td></tr>
<tr><td>AR</td><td>MAE ↓</td><td>0.5929</td><td><strong>0.5753</strong></td></tr>
</tbody>
</table></div>
<p class="note">Local evaluation using the same fixed procedure for ORIS BERT and PolBERTa base; not an official KLEJ leaderboard submission.</p>
</div></section>
<section><div class="wrap">
<div class="eyebrow">Document filtering</div>
<h3 class="section-title">The workload it was originally built for is where the design makes the most sense.</h3>
<div class="table-wrap"><table>
<thead><tr><th>Metric</th><th>mmBERT-base</th><th>ORIS BERT</th></tr></thead>
<tbody>
<tr class="highlight"><td>Decision Macro-F1</td><td>0.4334</td><td><strong>0.5015</strong></td></tr>
<tr><td>Decision accuracy</td><td>0.5185</td><td><strong>0.6296</strong></td></tr>
<tr><td>Training time</td><td>583.2 s</td><td><strong>124.4 s</strong></td></tr>
<tr><td>Peak VRAM</td><td>5.83 GiB</td><td><strong>0.52 GiB</strong></td></tr>
</tbody>
</table></div>
<h4 style="margin-top:32px">Production-style full pipeline</h4>
<div class="table-wrap"><table>
<thead><tr><th>Metric</th><th>mmBERT-base</th><th>ORIS BERT</th></tr></thead>
<tbody>
<tr class="highlight"><td>Full pipeline time</td><td>21.732 s</td><td><strong>4.408 s</strong></td></tr>
<tr><td>Documents / second</td><td>11.78</td><td><strong>58.07</strong></td></tr>
<tr><td>Mean latency / document</td><td>84.89 ms</td><td><strong>17.22 ms</strong></td></tr>
<tr><td>Peak VRAM</td><td>1.806 GiB</td><td><strong>0.252 GiB</strong></td></tr>
</tbody>
</table></div>
<div class="callout"><strong>~4.93× more documents per second in this workload.</strong><p>This is a task-specific pipeline result, not a claim that Small C is universally superior to larger encoders.</p></div>
</div></section>
<section><div class="wrap">
<div class="eyebrow">Encoder efficiency</div>
<h3 class="section-title">The compact architecture also showed a large raw forward-pass advantage.</h3>
<div class="table-wrap"><table>
<thead><tr><th>Setting</th><th>ORIS BERT</th><th>mmBERT-small</th><th>Advantage</th></tr></thead>
<tbody>
<tr><td>Batch 1, 128 tokens</td><td><strong>3.828 ms</strong></td><td>17.153 ms</td><td><strong>4.48×</strong></td></tr>
<tr><td>Batch 1, 1024 tokens</td><td><strong>3.940 ms</strong></td><td>17.229 ms</td><td><strong>4.37×</strong></td></tr>
<tr class="highlight"><td>Batch 8, 1024 tokens</td><td><strong>1.25M tok/s</strong></td><td>159K tok/s</td><td><strong>7.85×</strong></td></tr>
</tbody>
</table></div>
<p class="note">Different tokenizers mean cross-model tokens/s should be interpreted carefully.</p>
</div></section>
<section><div class="wrap">
<div class="eyebrow">Roadmap and limitations</div>
<h3 class="section-title">Small C is useful precisely because its limits are explicit.</h3>
<div class="prose">
<p>Raw mean-pooled sentence representations have poor cosine-space separation. Retrieval, ranking and sentence similarity need additional fine-tuning. Larger Polish encoders remain stronger on several benchmarks, and five local-attention layers mean cross-window communication relies heavily on one global mixing layer.</p>
<p>The model is therefore best understood as a compact MLM-pretrained backbone for supervised tasks, filtering and feature extraction — not as a universal zero-shot embedding model.</p>
</div>
<div class="timeline">
<div class="timeline-item"><div class="timeline-year">Origin</div><div class="timeline-body"><h4>ORIS 660M pipeline bottleneck</h4><p>Built to make local filtering and scoring fast enough to matter.</p></div></div>
<div class="timeline-item"><div class="timeline-year">16 Aug 2026</div><div class="timeline-body"><h4>Small C checkpoint completed</h4><p>8.00B-token from-scratch pretraining completed.</p></div></div>
<div class="timeline-item"><div class="timeline-year">Current</div><div class="timeline-body"><h4>Gated research release</h4><p>Used as a practical encoder backbone while downstream and representation-space behaviour are evaluated.</p></div></div>
<div class="timeline-item"><div class="timeline-year">Possible next</div><div class="timeline-body"><h4>Broader ORIS encoder line</h4><p>Further architecture, training and evaluation work may turn the concept-stage checkpoint into a fuller encoder family.</p></div></div>
</div>
</div></section>
</article>
<article class="detail" id="model-vyuhu">
<div class="detail-top"><div class="wrap detail-bar">
<button class="back" onclick="showIndex()">← All models</button>
<div class="detail-name">Vyuhu</div>
</div></div>
<div class="wrap detail-hero">
<div class="eyebrow">Active architecture research</div>
<h2>Vyuhu</h2>
<p>
<strong>One shot, one compute, many possibilities.</strong>
One jointly trained model exposes four deterministic compute profiles and can later be physically reduced into smaller standalone models
without changing what a given profile computes.
</p>
<div class="stats">
<div class="stat"><strong>4 + 1</strong><span>four compute profiles + the shared supernetwork</span></div>
<div class="stat"><strong>0.834B</strong><span>tokens in the completed architecture-validation run</span></div>
<div class="stat"><strong>Δlogit = 0</strong><span>exact profile-vs-extracted-model equivalence</span></div>
<div class="stat"><strong>~11.8k tok/s</strong><span>~500M smoke / training-path throughput on RTX 5060 Ti 16 GB</span></div>
</div>
<div class="update-log">
<div class="update-item" onclick="scrollInsideModel('vyuhu-proof')">
<div class="update-date">validated</div>
<div class="update-copy"><strong>0.834B-token architecture run</strong><span>Compute ordering, profile separation and exact physical extraction all passed the intended checks.</span></div>
</div>
<div class="update-item" onclick="scrollInsideModel('vyuhu-family')">
<div class="update-date">next run</div>
<div class="update-copy"><strong>~280M family target</strong><span>A hardware-friendly first full run aimed at roughly 280M / 175M / 125M / 100M operating points.</span></div>
</div>
<div class="update-item" onclick="scrollInsideModel('vyuhu-500m')">
<div class="update-date">smoke</div>
<div class="update-copy"><strong>~500M scale test</strong><span>The architecture and training path were also instantiated at ~501.8M parameters and sustained roughly 11.8k tok/s in the current optimized path.</span></div>
</div>
</div>
</div>
<section id="vyuhu-concept"><div class="wrap">
<div class="eyebrow">The idea</div>
<h3 class="section-title">Not MoE. Not token routing. One deterministic family inside one trained network.</h3>
<div class="prose">
<p>
Vyuhu does not learn a router that decides which experts receive each token. A profile is selected before the forward pass.
Mandatory anchor blocks are always present; optional heavy compute is executed according to a fixed profile schedule.
Cheap transfer paths and low-rank controllers keep representations compatible when compute is skipped.
</p>
<p>
The result is one training lineage with several explicit operating points:
<strong>Vasudeva</strong>, <strong>Sankarshana</strong>, <strong>Pradyumna</strong> and <strong>Aniruddha</strong>.
The fifth object in “4 in one? Nah, 5.” is the complete shared supernetwork from which those paths come.
</p>
</div>
<div class="version-grid">
<div class="version-card"><small>Vasudeva</small><h4>Full path</h4><p>The largest deterministic path. Nothing optional is withheld.</p></div>
<div class="version-card"><small>Sankarshana</small><h4>Middle-high compute</h4><p>A reduced path intended to land near the ~175M class in the first ~280M family run.</p></div>
<div class="version-card"><small>Pradyumna</small><h4>Middle-low compute</h4><p>A smaller operating point targeted near the ~125M class.</p></div>
<div class="version-card"><small>Aniruddha</small><h4>Anchor-first minimum</h4><p>The smallest deterministic path, targeted near ~100M in the first full family run.</p></div>
</div>
<div class="callout">
<strong>One shot, one compute, many possibilities.</strong>
<p>Train the shared system once. Select a deterministic compute level at inference — or extract that level into its own standalone model later.</p>
</div>
</div></section>
<section id="vyuhu-proof"><div class="wrap">
<div class="eyebrow">Architecture validation</div>
<h3 class="section-title">The 0.834B-token run passed every architectural check it was designed to answer.</h3>
<div class="prose">
<p>
The validation run was not treated as a final quality leaderboard. Its purpose was narrower and more important:
determine whether one jointly trained network could preserve a stable compute→quality ordering across deterministic profiles,
and whether a selected profile could be physically removed from the supernetwork without silently changing its function.
</p>
<p>
At roughly <strong>0.832B evaluated tokens</strong>, quality remained monotonic with compute:
the full Vasudeva path was best, followed by Sankarshana, Pradyumna and Aniruddha.
No profile inversion appeared in the final evaluation.
</p>
</div>
<div class="table-wrap"><table>
<thead><tr><th>Profile</th><th>Schedule</th><th>Physical extracted size</th><th>Eval loss</th><th>Perplexity</th></tr></thead>
<tbody>
<tr class="highlight"><td>Vasudeva</td><td>[4, 4, 4]</td><td>125.721M</td><td>3.2705</td><td>26.32</td></tr>
<tr><td>Sankarshana</td><td>[1, 2, 1]</td><td>75.602M</td><td>3.3174</td><td>27.59</td></tr>
<tr><td>Pradyumna</td><td>[0, 1, 0]</td><td>56.723M</td><td>3.4138</td><td>30.38</td></tr>
<tr><td>Aniruddha</td><td>[0, 0, 0]</td><td>50.430M</td><td>3.5446</td><td>34.63</td></tr>
</tbody>
</table></div>
<div class="stats">
<div class="stat"><strong>V > S > P > A</strong><span>monotonic compute-quality ordering</span></div>
<div class="stat"><strong>4 / 4</strong><span>profiles physically extracted</span></div>
<div class="stat"><strong>0.0</strong><span>maximum profile-vs-extracted logit delta in exact extraction test</span></div>
<div class="stat"><strong>100%</strong><span>of the targeted architecture checks passed in this run</span></div>
</div>
<div class="callout">
<strong>“Extraction” is not another approximation step.</strong>
<p>
The extracted model reproduces the selected profile exactly: <strong>Δlogit = 0</strong> in the extraction test.
A smaller extracted artifact can therefore be deployed without an additional quality loss caused by the act of removing inactive structure.
The profiles themselves still have different quality because they intentionally use different amounts of compute.
</p>
</div>
</div></section>
<section id="vyuhu-family"><div class="wrap">
<div class="eyebrow">First full run</div>
<h3 class="section-title">A ~280M supernetwork that behaves like a small Polish model family.</h3>
<div class="prose">
<p>
The next target is deliberately smaller than the 500M engineering scale test.
Width is reduced while the proven four-profile topology is retained, with the middle profiles raised by one heavy block
so the family lands near four useful deployment classes rather than clustering too tightly at the bottom.
</p>
</div>
<div class="table-wrap"><table>
<thead><tr><th>Profile</th><th>Planned schedule</th><th>Approx. family class</th><th>Role</th></tr></thead>
<tbody>
<tr class="highlight"><td>Vasudeva</td><td>[4, 4, 4]</td><td>~275–285M</td><td>full model</td></tr>
<tr><td>Sankarshana</td><td>[1, 3, 1]</td><td>~175M</td><td>balanced middle-high path</td></tr>
<tr><td>Pradyumna</td><td>[0, 2, 0]</td><td>~125M</td><td>compact middle-low path</td></tr>
<tr><td>Aniruddha</td><td>[0, 0, 0]</td><td>~100M</td><td>minimum anchor path</td></tr>
</tbody>
</table></div>
<p class="note">Sizes are current design targets for the first ~280M run, not yet final checkpoint counts.</p>
<div class="version-grid">
<div class="version-card"><small>Tokenizer</small><h4>VYUHU32k</h4><p>32K corpus-specific byte-level tokenizer prepared for the Polish training corpus and exact Unicode round-trip behaviour.</p></div>
<div class="version-card"><small>Hidden width</small><h4>1152</h4><p>Hardware-friendly width with 64-dimensional attention heads.</p></div>
<div class="version-card"><small>Attention</small><h4>GQA anchors</h4><p>18 query heads / 6 KV heads in the current ~280M design. Optional heavy blocks use the elastic mixer rather than full global attention.</p></div>
<div class="version-card"><small>FFN</small><h4>3584 · SwiGLU</h4><p>Dense SwiGLU channel capacity retained across anchor and elastic compute.</p></div>
<div class="version-card"><small>Optimizer</small><h4>Muon + fused AdamW</h4><p>Hybrid optimizer path: Muon for selected 2D hidden matrices, fused AdamW for the remaining parameter groups.</p></div>
<div class="version-card"><small>Some internals</small><h4>:)</h4><p>Controller/bypass details, training mixture and a few run-level knobs stay intentionally unpublished until the experiment is complete.</p></div>
</div>
</div></section>
<section id="vyuhu-reference"><div class="wrap">
<div class="eyebrow">Scale reference</div>
<h3 class="section-title">The first Vyuhu run sits in the same broad size class as a conventional ~275M MHA-style Polish LM — but spends parameters differently.</h3>
<div class="prose">
<p>
A useful conventional reference is the published ~275M configuration shown below.
It is not presented as a controlled Vyuhu benchmark: the architectures, optimizer path, tokenizer and training data differ.
The point is simply to make the scale tangible before the first ~280M Vyuhu run exists.
</p>
</div>
<div class="table-wrap"><table>
<thead><tr><th>Hyperparameter</th><th>Conventional ~275M reference</th><th>Vyuhu ~280M target</th></tr></thead>
<tbody>
<tr><td>Model parameters</td><td>275M</td><td>~275–285M supernetwork target</td></tr>
<tr><td>Sequence length</td><td>1024</td><td>1024</td></tr>
<tr><td>Vocabulary</td><td>31,980</td><td>32,000 · VYUHU32k</td></tr>
<tr><td>Transformer layers</td><td>32</td><td>4 mandatory GQA anchors + elastic staged compute</td></tr>
<tr><td>Attention heads</td><td>16 MHA</td><td>18 Q / 6 KV in GQA anchors</td></tr>
<tr><td>Head dimension</td><td>64</td><td>64</td></tr>
<tr><td>Model width</td><td>768</td><td>1152</td></tr>
<tr><td>Intermediate size</td><td>2048</td><td>3584</td></tr>
<tr><td>Positional encoding</td><td>RoPE</td><td>RoPE</td></tr>
<tr><td>Activation</td><td>SwiGLU</td><td>SwiGLU</td></tr>
<tr><td>Normalization</td><td>RMSNorm · ε 1e-6</td><td>RMSNorm · ε 1e-6</td></tr>
<tr><td>Dropout / bias</td><td>0.0 / no</td><td>0.0 / no</td></tr>
<tr><td>Optimizer</td><td>AdamW</td><td><strong>Muon + fused AdamW</strong></td></tr>
<tr><td>Exact MB / GA / LR schedule</td><td>13 / 40 / 4e-4 → 2e-5</td><td>:)</td></tr>
</tbody>
</table></div>
<div class="callout">
<strong>Different objective, not just a strange way to build another 280M model.</strong>
<p>
A conventional dense 275M checkpoint is one operating point. Vyuhu is being designed so a single training run can yield
a ~280M full path and useful ~175M, ~125M and ~100M deterministic descendants — with exact extraction rather than post-hoc pruning.
</p>
</div>
</div></section>
<section id="vyuhu-500m"><div class="wrap">
<div class="eyebrow">Scale smoke test</div>
<h3 class="section-title">The design was also instantiated at ~501.8M parameters before committing to the smaller final run.</h3>
<div class="prose">
<p>
The ~500M branch was used as an engineering stress test for memory, optimizer geometry, profiling and custom-kernel work on a single RTX 5060 Ti 16 GB.
The current training path reached roughly <strong>11.8k tokens/s</strong> at sequence length 1024 with MB 8 / GA 16.
</p>
<p>
This is a throughput smoke test, not a quality claim for a finished 500M checkpoint.
It demonstrated that the architecture, hybrid optimizer and current runtime can execute at this scale on the target desktop GPU.
</p>
</div>
<div class="stats">
<div class="stat"><strong>501.83M</strong><span>engineering-scale parameter count</span></div>
<div class="stat"><strong>1024</strong><span>sequence length</span></div>
<div class="stat"><strong>MB 8 / GA 16</strong><span>measured training geometry</span></div>
<div class="stat"><strong>~11.8k tok/s</strong><span>optimized full training path</span></div>
</div>
</div></section>
<section id="vyuhu-riddle"><div class="wrap">
<div class="eyebrow">The riddle survived</div>
<h3 class="section-title">4 in one? Nah, 5.</h3>
<div class="version-grid">
<div class="version-card"><small>Vyuhu</small><h4>What are you?</h4>
<p><strong>Vasudeva:</strong> The form that answers when nothing needs to be left behind.</p>
<p><strong>Sankarshana:</strong> The same answer, after deciding some of itself can remain silent.</p>
<p><strong>Pradyumna:</strong> What appears when less is allowed to be enough.</p>
<p><strong>Aniruddha:</strong> The version that still reaches the answer after forgetting how much it was supposed to carry.</p>
</div>
<div class="version-card"><small>Vyuhu</small><h4>Which one of you is Vyuhu?</h4>
<p><strong>Vasudeva:</strong> I.</p>
<p><strong>Sankarshana:</strong> I.</p>
<p><strong>Pradyumna:</strong> I.</p>
<p><strong>Aniruddha:</strong> I.</p>
<p><strong>Supernetwork:</strong> You trained me once.</p>
</div>
</div>
<p class="note">Vyuhu · active architecture research · first ~280M full run in preparation</p>
</div></section>
</article>
<article class="detail" id="model-vidar">
<div class="detail-top"><div class="wrap detail-bar">
<button class="back" onclick="showIndex()">← All models</button>
<div class="detail-name">Vidar-VL</div>
</div></div>
<div class="wrap detail-hero">
<div class="eyebrow">Planned vision-language research</div>
<h2>Vidar-VL</h2>
<p>A planned ORIS vision-language model focused on understanding images and interacting with visual content in Polish.</p>
<div class="chips">
<span class="chip">image understanding</span>
<span class="chip">visual question answering</span>
<span class="chip">OCR / documents</span>
<span class="chip">Polish visual interaction</span>
<span class="chip">multimodal reasoning</span>
</div>
</div>
<section><div class="wrap">
<div class="eyebrow">Status</div>
<h3 class="section-title">Planned and currently under development.</h3>
<div class="prose">
<p>Vidar-VL is intended as the ORIS vision-language branch for image understanding, visual question answering, OCR-oriented interaction and broader multimodal reasoning.</p>
</div>
<p class="note">Further technical details will be published when the project reaches its first experimental release.</p>
</div></section>
</article>
<article class="detail" id="model-orisvision">
<div class="detail-top"><div class="wrap detail-bar">
<button class="back" onclick="showIndex()">← All models</button>
<div class="detail-name">ORIS Vision</div>
</div></div>
<div class="wrap detail-hero">
<div class="eyebrow">Planned visual representation research</div>
<h2>ORIS Vision</h2>
<p>A planned compact visual embedding model — the visual counterpart to ORIS BERT — for similarity, retrieval, scoring, analysis and filtering.</p>
<div class="chips">
<span class="chip">visual embeddings</span>
<span class="chip">image scoring</span>
<span class="chip">similarity</span>
<span class="chip">retrieval</span>
<span class="chip">quality filtering</span>
</div>
</div>
<section><div class="wrap">
<div class="eyebrow">Status</div>
<h3 class="section-title">The visual half of the compact embedding-model pair.</h3>
<div class="prose">
<p>ORIS Vision is intended as the visual counterpart to ORIS BERT: both are compact encoder-style models aimed at representations rather than conversational generation. Planned uses include image embeddings, similarity, retrieval, scoring and image filtering.</p>
</div>
<p class="note">Architecture, training sources and implementation details are intentionally not disclosed at this stage.</p>
</div></section>
</article>
<footer><div class="wrap">ORIS Research · Independent experimental model work · 2024–2027</div></footer>
<script>
const ids=["polmath","oris660","smallc","vyuhu","vidar","orisvision"];
function hideDetails(){
ids.forEach(id=>{
const el=document.getElementById("model-"+id);
if(el) el.classList.remove("active");
});
}
function showModel(id){
const model=document.getElementById("model-"+id);
if(!model) return;
document.getElementById("index").style.display="none";
hideDetails();
model.classList.add("active");
history.replaceState(null,"","#"+id);
window.scrollTo(0,0);
}
function showIndex(){
hideDetails();
document.getElementById("index").style.display="block";
history.replaceState(null,"","#models");
window.scrollTo(0,0);
}
function scrollInsideModel(targetId){
const target=document.getElementById(targetId);
if(!target) return;
target.scrollIntoView({behavior:"smooth",block:"start"});
}
window.addEventListener("DOMContentLoaded",()=>{
const h=location.hash.slice(1);
if(ids.includes(h)) showModel(h); else showIndex();
});
</script>
</body>
</html> |