File size: 31,400 Bytes
930e291
 
 
 
 
d24dca8
930e291
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
d24dca8
930e291
 
 
 
 
 
 
d24dca8
 
930e291
 
 
 
 
 
 
d24dca8
930e291
 
d24dca8
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
930e291
d24dca8
930e291
d24dca8
 
930e291
 
 
 
 
 
 
 
d24dca8
930e291
 
d24dca8
 
 
930e291
 
 
 
d24dca8
 
 
 
 
 
 
 
 
930e291
 
d24dca8
 
930e291
 
 
d24dca8
 
 
930e291
 
d24dca8
930e291
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
d24dca8
930e291
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
d24dca8
930e291
 
 
 
 
 
 
 
 
 
 
 
 
533b008
 
 
 
 
 
124f5a3
533b008
 
 
 
 
 
 
 
 
 
 
 
 
d24dca8
533b008
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
d24dca8
533b008
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
930e291
 
 
 
 
 
 
 
 
d24dca8
930e291
 
d24dca8
930e291
 
 
 
 
 
d24dca8
 
930e291
 
 
 
 
d24dca8
 
930e291
 
 
 
 
d24dca8
 
930e291
 
 
 
 
d24dca8
 
930e291
 
 
 
 
d24dca8
 
930e291
 
 
d24dca8
930e291
d24dca8
 
930e291
 
 
 
 
613243b
930e291
 
 
 
 
 
 
 
 
 
d24dca8
930e291
 
 
 
 
 
 
 
 
 
72148e6
 
930e291
 
 
 
 
d24dca8
 
930e291
 
 
 
 
 
 
72148e6
db777f7
72148e6
930e291
 
72148e6
 
d24dca8
930e291
 
 
 
 
 
 
 
72148e6
930e291
72148e6
930e291
 
 
 
 
 
67081eb
72148e6
930e291
 
 
 
02e9bc2
 
 
 
 
 
 
 
d24dca8
67081eb
930e291
 
96c2919
930e291
 
 
d24dca8
 
930e291
 
 
ae650e1
 
 
d24dca8
ae650e1
d24dca8
 
ae650e1
930e291
 
ae650e1
 
 
d24dca8
b7d4c7a
ae650e1
d24dca8
ae650e1
 
d24dca8
ae650e1
930e291
d24dca8
 
 
930e291
 
 
d24dca8
 
930e291
 
 
d24dca8
ae650e1
930e291
d24dca8
 
 
930e291
 
 
d24dca8
930e291
d24dca8
 
 
 
 
 
ae650e1
d24dca8
 
 
 
 
ae650e1
 
930e291
d24dca8
ae650e1
 
d24dca8
 
 
ae650e1
 
 
d24dca8
 
 
 
 
 
 
 
 
 
 
 
930e291
 
d24dca8
930e291
 
 
 
 
 
 
 
 
d24dca8
 
930e291
 
 
 
d24dca8
 
930e291
 
d24dca8
 
930e291
 
d24dca8
 
930e291
 
d24dca8
 
930e291
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
d24dca8
 
930e291
 
 
 
d24dca8
 
930e291
 
 
 
d24dca8
 
930e291
 
 
 
d24dca8
 
930e291
 
d24dca8
930e291
d24dca8
 
930e291
 
 
 
613243b
930e291
 
 
 
 
 
 
 
6e85609
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
<!DOCTYPE html>
<html lang="en">
<head>
  <meta charset="UTF-8">
  <meta name="viewport" content="width=device-width, initial-scale=1.0">
  <title>Kalpana AI โ€” O(1) RIF Neural Studio & Benchmarks</title>
  
  <!-- Fonts -->
  <link rel="preconnect" href="https://fonts.googleapis.com">
  <link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
  <link href="https://fonts.googleapis.com/css2?family=JetBrains+Mono:wght@400;500;600;700&family=Outfit:wght@300;400;500;600;700;800&display=swap" rel="stylesheet">
  
  <!-- KaTeX for formulas -->
  <link rel="stylesheet" href="https://cdn.jsdelivr.net/npm/katex@0.16.8/dist/katex.min.css">
  <script defer src="https://cdn.jsdelivr.net/npm/katex@0.16.8/dist/katex.min.js"></script>

  <link rel="stylesheet" href="./style.css">
</head>
<body>
  <div class="app-layout">
    <!-- Navigation Top Bar -->
    <header class="top-nav">
      <div class="brand">
        <div class="logo-badge">K</div>
        <div>
          <div class="logo-title">Kalpanฤ AI Studio</div>
          <div class="logo-sub">O(1) Resonant Interference Field Substrate ยท Patent LK/P/1/24089</div>
        </div>
      </div>

      <nav class="nav-tabs">
        <button class="nav-tab active" data-tab="tab-chat">๐Ÿ’ฌ Live Neural Chat</button>
        <button class="nav-tab" data-tab="tab-benchmark">๐Ÿ”ฌ Benchmarks & Haystack</button>
        <button class="nav-tab" data-tab="tab-architecture">๐Ÿ›๏ธ Layer Architecture</button>
        <button class="nav-tab" data-tab="tab-swagger">๐Ÿ”Œ Swagger API</button>
        <button class="nav-tab" data-tab="tab-economics">๐Ÿ’ฐ Unit Economics</button>
      </nav>

      <div class="header-status">
        <span class="status-indicator" id="headerStatusDot"></span>
        <span id="headerStatusText">O(1) RIF GPU Active (96.00 MB ยท 24 Layers)</span>
      </div>
    </header>

    <!-- Main Content Container -->
    <div class="tab-content-container">

      <!-- ============================================================ -->
      <!-- TAB 1: LIVE NEURAL CHAT (FULL WIDTH CLEAN INTERFACE) -->
      <!-- ============================================================ -->
      <section class="tab-pane active" id="tab-chat">
        <div class="chat-container">
          
          <!-- Neural GPU Telemetry & Health Bar -->
          <div class="chat-telemetry-bar">
            <div class="telemetry-item">
              <span class="pulse-dot" id="serverPulse"></span>
              <span class="telemetry-label">GPU Backend:</span>
              <span class="telemetry-val val-green" id="serverStatusVal">NVIDIA GPU ยท Online</span>
            </div>
            <div class="telemetry-item">
              <span class="telemetry-label">Attention Routing:</span>
              <span class="telemetry-val val-cyan">24 / 24 Layers Intercepted</span>
            </div>
            <div class="telemetry-item">
              <span class="telemetry-label">O(1) KV Memory:</span>
              <span class="telemetry-val val-green">96.00 MB (Strict O(1))</span>
            </div>
            <div class="telemetry-item">
              <span class="telemetry-label">Harmonic Bands:</span>
              <span class="telemetry-val val-purple">2,048 Bands</span>
            </div>
            <button class="btn-ping" id="btnPingServer" title="Test real-time connection to GPU backend">
              ๐Ÿ”„ Ping Server
            </button>
          </div>

          <!-- Chat History Stream -->
          <main class="chat-main-full">
            <div class="chat-history" id="chatHistory">
              <div class="chat-bubble bot-bubble">
                <div class="bubble-header">
                  <span class="bubble-avatar">K</span>
                  <span class="bubble-author">Kalpana AI</span>
                  <span class="bubble-badge">Qwen2.5-0.5B + RIF</span>
                </div>
                <div class="bubble-body">
                  Hello! ๐Ÿ‘‹ I am **Kalpana AI**, powered by the **Qwen2.5-0.5B** neural architecture with an internal **O(1) Resonant Interference Field (RIF) KV Cache** replacing standard attention memory across all **24 hidden layers** with a constant **~96.00 MB VRAM** footprint ($O(1)$ invariant).
                  
                  How can I help you today?
                  - Ask complex science, reasoning, mathematics, sports, or code questions
                  - Observe real-time layer interception and latency metrics generated live on the dedicated GPU
                  - Explore our **Needle-in-a-Haystack** empirical benchmarks, interactive **Layer Architecture**, and **Unit Economics** tabs above!
                </div>
              </div>
            </div>

            <!-- Generating Progress Indicator Bar -->
            <div class="gen-progress-bar" id="genProgressBar" style="display: none;">
              <div class="progress-track">
                <div class="progress-fill"></div>
              </div>
              <div class="progress-text">โšก Routing prompt through 24 RIF Attention Layers on GPU...</div>
            </div>

            <!-- Input Bar -->
            <div class="chat-input-wrapper">
              <div class="chat-input-bar">
                <textarea id="chatInput" placeholder="Ask anything, test math, physics, reasoning, or code... (Press Enter to Send)" rows="1"></textarea>
                <button id="btnSendChat" class="btn-send" title="Send query to Kalpana RIF Engine">
                  <svg width="18" height="18" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2.5"><line x1="22" y1="2" x2="11" y2="13"/><polygon points="22 2 15 22 11 13 2 9 22 2"/></svg>
                </button>
              </div>
              <div class="input-caption">
                Direct neural forward-pass through <code>KalpanaDynamicCache</code> on dedicated NVIDIA GPU ยท Strict O(1) Memory Invariance.
              </div>
            </div>
          </main>

        </div>
      </section>

      <!-- ============================================================ -->
      <!-- TAB 2: BENCHMARKS & NEEDLE-IN-A-HAYSTACK -->
      <!-- ============================================================ -->
      <section class="tab-pane" id="tab-benchmark">
        <div class="pane-inner">
          <div class="section-header">
            <h2>๐Ÿ”ฌ Empirical Retrieval & Memory Benchmarks</h2>
            <p>Evaluating long-context recall across 500 semantic chunks (~12,500 tokens) and memory scaling bounds.</p>
          </div>

          <!-- Needle in Haystack Live Runner -->
          <div class="content-card">
            <div class="card-head">
              <h3>๐ŸŽฏ Needle-in-a-Haystack Test Suite (500 Chunks / 2,048 Bands)</h3>
              <button class="btn-primary" id="btnRunHaystack" style="width: auto; padding: 0.5rem 1.2rem;">
                โ–ถ Run Live Test Suite
              </button>
            </div>

            <div class="benchmark-grid">
              <div class="haystack-card" id="needle1Card">
                <div class="needle-badge">NEEDLE 1 ยท 10% DEPTH (t=50)</div>
                <div class="needle-query">"What is the secret passkey for Project Chronos?"</div>
                <div class="needle-result">
                  <span class="status-tag tag-pass">EXACT HIT (Resonance: 0.8935)</span>
                  <div class="retrieved-text">"The secret passkey for Project Chronos is OMEGA-7749."</div>
                </div>
              </div>

              <div class="haystack-card" id="needle2Card">
                <div class="needle-badge">NEEDLE 2 ยท 50% DEPTH (t=250)</div>
                <div class="needle-query">"Who invented the resonant hyper-drive?"</div>
                <div class="needle-result">
                  <span class="status-tag tag-pass">EXACT HIT (Resonance: 0.7880)</span>
                  <div class="retrieved-text">"Dr. Elena Vance invented the resonant hyper-drive in Neo-Geneva."</div>
                </div>
              </div>

              <div class="haystack-card" id="needle3Card">
                <div class="needle-badge">NEEDLE 3 ยท 90% DEPTH (t=450)</div>
                <div class="needle-query">"What is the emergency shutdown code for reactor 4?"</div>
                <div class="needle-result">
                  <span class="status-tag tag-pass">EXACT HIT (Resonance: 0.8293)</span>
                  <div class="retrieved-text">"The emergency shutdown code for reactor 4 is EPSILON-9021."</div>
                </div>
              </div>
            </div>

            <div class="stats-banner">
              <div class="stat-box">
                <div class="stat-number">100.0%</div>
                <div class="stat-label">Retrieval Accuracy (3/3 Exact Hits)</div>
              </div>
              <div class="stat-box">
                <div class="stat-number">96.00 MB</div>
                <div class="stat-label">Active Memory Footprint (Strict O(1))</div>
              </div>
              <div class="stat-box">
                <div class="stat-number">20.2</div>
                <div class="stat-label">Ingestion Speed (chunks / sec)</div>
              </div>
              <div class="stat-box">
                <div class="stat-number">0.00 ms</div>
                <div class="stat-label">Prompt Re-Transmission Overhead</div>
              </div>
            </div>
          </div>

          <!-- โš”๏ธ Live Head-to-Head Benchmark Suite -->
          <div class="content-card" style="margin-top: 1.5rem; border-color: rgba(124, 58, 237, 0.4);">
            <div class="card-head">
              <div>
                <h3 style="color: #c084fc;">โš”๏ธ Live Head-to-Head Benchmark: Baseline Qwen vs. Kalpana RIF Qwen</h3>
                <p style="font-size: 0.82rem; color: var(--text-muted); margin-top: 0.2rem;">
                  Real-time tensor footprint comparison across expanding token horizons (2K to 1M tokens) based on verified PyTorch attention equations.
                </p>
              </div>
              <button class="btn-primary" id="btnRunH2H" style="width: auto; padding: 0.5rem 1.2rem; background: linear-gradient(135deg, #7c3aed, #00f0ff);">
                โ–ถ Run Live Head-to-Head Test
              </button>
            </div>

            <!-- Dynamic Comparison Columns -->
            <div style="display: grid; grid-template-columns: repeat(auto-fit, minmax(320px, 1fr)); gap: 1.2rem; margin-top: 1rem;">
              <!-- Model A: Baseline Qwen (Standard KV Cache) -->
              <div style="background: rgba(255, 51, 102, 0.04); border: 1px solid rgba(255, 51, 102, 0.3); border-radius: 10px; padding: 1.2rem;">
                <div style="display: flex; justify-content: space-between; align-items: center; margin-bottom: 0.8rem;">
                  <span style="font-weight: 700; color: var(--red); font-size: 0.95rem;">๐Ÿšซ Baseline Qwen (Standard KV Cache)</span>
                  <span class="status-tag tag-fail" id="baselineStatusTag">O(N) Linear Growth</span>
                </div>
                <div style="font-size: 0.8rem; color: var(--text-muted); margin-bottom: 1rem;">
                  Tensor scaling: <code>torch.cat([cache, new_kv], dim=-2)</code> across all 24 layers.
                </div>

                <div style="display: flex; flex-direction: column; gap: 0.7rem;">
                  <div>
                    <div style="display: flex; justify-content: space-between; font-size: 0.8rem; margin-bottom: 0.2rem;">
                      <span style="color: var(--text-secondary);">Active Context:</span>
                      <strong id="h2hBaseTokens" style="font-family: var(--font-mono); color: #fff;">0 tokens</strong>
                    </div>
                    <div style="display: flex; justify-content: space-between; font-size: 0.8rem; margin-bottom: 0.2rem;">
                      <span style="color: var(--text-secondary);">KV Cache Memory:</span>
                      <strong id="h2hBaseMemory" style="font-family: var(--font-mono); color: var(--red);">0.00 MB</strong>
                    </div>
                    <div style="display: flex; justify-content: space-between; font-size: 0.8rem; margin-bottom: 0.4rem;">
                      <span style="color: var(--text-secondary);">Latency per Token:</span>
                      <strong id="h2hBaseLatency" style="font-family: var(--font-mono); color: var(--red);">-- ms</strong>
                    </div>
                    <div style="background: rgba(0,0,0,0.5); border-radius: 4px; height: 10px; overflow: hidden; border: 1px solid rgba(255,51,102,0.2);">
                      <div id="h2hBaseBar" style="background: linear-gradient(90deg, #ff9900, #ff3366); height: 100%; width: 0%; transition: width 0.3s ease;"></div>
                    </div>
                  </div>
                  <div id="h2hBaseAlert" style="font-size: 0.78rem; padding: 0.5rem; background: rgba(0,0,0,0.4); border-radius: 6px; color: var(--text-muted); min-height: 2.2rem;">
                    Ready to run benchmark.
                  </div>
                </div>
              </div>

              <!-- Model B: Kalpana RIF Qwen (O(1) Dynamic Cache) -->
              <div style="background: rgba(0, 255, 136, 0.04); border: 1px solid rgba(0, 255, 136, 0.3); border-radius: 10px; padding: 1.2rem;">
                <div style="display: flex; justify-content: space-between; align-items: center; margin-bottom: 0.8rem;">
                  <span style="font-weight: 700; color: var(--green); font-size: 0.95rem;">โšก Kalpana RIF Qwen (DynamicCache)</span>
                  <span class="status-tag tag-pass" id="kalpanaStatusTag">O(1) Invariant</span>
                </div>
                <div style="font-size: 0.8rem; color: var(--text-muted); margin-bottom: 1rem;">
                  Wave interference: <code>KalpanaCacheLayer.write()</code> across all 24 layers.
                </div>

                <div style="display: flex; flex-direction: column; gap: 0.7rem;">
                  <div>
                    <div style="display: flex; justify-content: space-between; font-size: 0.8rem; margin-bottom: 0.2rem;">
                      <span style="color: var(--text-secondary);">Active Context:</span>
                      <strong id="h2hKalpTokens" style="font-family: var(--font-mono); color: #fff;">0 tokens</strong>
                    </div>
                    <div style="display: flex; justify-content: space-between; font-size: 0.8rem; margin-bottom: 0.2rem;">
                      <span style="color: var(--text-secondary);">KV Cache Memory:</span>
                      <strong id="h2hKalpMemory" style="font-family: var(--font-mono); color: var(--green);">96.00 MB (Strict O(1))</strong>
                    </div>
                    <div style="display: flex; justify-content: space-between; font-size: 0.8rem; margin-bottom: 0.4rem;">
                      <span style="color: var(--text-secondary);">Latency per Token:</span>
                      <strong id="h2hKalpLatency" style="font-family: var(--font-mono); color: var(--green);">-- ms</strong>
                    </div>
                    <div style="background: rgba(0,0,0,0.5); border-radius: 4px; height: 10px; overflow: hidden; border: 1px solid rgba(0,255,136,0.2);">
                      <div id="h2hKalpBar" style="background: linear-gradient(90deg, #00f0ff, #00ff88); height: 100%; width: 5%; transition: width 0.3s ease;"></div>
                    </div>
                  </div>
                  <div id="h2hKalpAlert" style="font-size: 0.78rem; padding: 0.5rem; background: rgba(0,0,0,0.4); border-radius: 6px; color: var(--green); min-height: 2.2rem;">
                    Ready to run benchmark.
                  </div>
                </div>
              </div>
            </div>
          </div>

          <!-- Memory Scaling Comparison Table -->
          <div class="content-card" style="margin-top: 1.5rem;">
            <div class="card-head">
              <h3>๐Ÿ“Š Memory Scaling Comparison: Standard Linear KV Cache vs. Kalpana RIF</h3>
            </div>
            <table class="data-table">
              <thead>
                <tr>
                  <th>Context Horizon</th>
                  <th>Standard KV Cache (Qwen2.5 / Llama-3)</th>
                  <th>Kalpana RIF (O(1))</th>
                  <th>Memory Reduction</th>
                  <th>Status on Single GPU</th>
                </tr>
              </thead>
              <tbody>
                <tr>
                  <td><strong>2,000 tokens</strong></td>
                  <td>256 MB</td>
                  <td><strong class="val-good">96.00 MB</strong></td>
                  <td>2.7ร— smaller</td>
                  <td><span class="tag-pass">Fits</span></td>
                </tr>
                <tr>
                  <td><strong>8,000 tokens</strong></td>
                  <td>1,024 MB (1.0 GB)</td>
                  <td><strong class="val-good">96.00 MB</strong></td>
                  <td>10.6ร— smaller</td>
                  <td><span class="tag-pass">Fits</span></td>
                </tr>
                <tr>
                  <td><strong>32,000 tokens</strong></td>
                  <td>4,096 MB (4.0 GB)</td>
                  <td><strong class="val-good">96.00 MB</strong></td>
                  <td>42.6ร— smaller</td>
                  <td><span class="tag-pass">Fits</span></td>
                </tr>
                <tr>
                  <td><strong>128,000 tokens</strong></td>
                  <td>16,384 MB (16.0 GB)</td>
                  <td><strong class="val-good">96.00 MB</strong></td>
                  <td>170ร— smaller</td>
                  <td><span class="tag-warn">High VRAM Strain</span></td>
                </tr>
                <tr>
                  <td><strong>1,000,000 tokens</strong></td>
                  <td>138,000 MB (138 GB)</td>
                  <td><strong class="val-good">96.00 MB</strong></td>
                  <td><strong>1,437ร— smaller</strong></td>
                  <td><span class="tag-fail">โŒ Out Of Memory (OOM)</span></td>
                </tr>
                <tr>
                  <td><strong>3,000,000 tokens</strong></td>
                  <td>384,000 MB (384 GB)</td>
                  <td><strong class="val-good">96.00 MB</strong></td>
                  <td><strong>4,000ร— smaller</strong></td>
                  <td><span class="tag-fail">โŒ Needs 5ร— A100 GPUs</span></td>
                </tr>
              </tbody>
            </table>
          </div>

        </div>
      </section>

      <!-- ============================================================ -->
      <!-- TAB 3: LAYER ARCHITECTURE & LLM INTERCEPTION -->
      <!-- ============================================================ -->
      <section class="tab-pane" id="tab-architecture">
        <div class="pane-inner">
          <div class="section-header">
            <h2>๐Ÿ›๏ธ Deep LLM Layer Architecture: Where RIF Intercepts Attention</h2>
            <p>How Kalpana replaces unbounded tensor concatenation (`torch.cat`) with continuous wave interference across all 24 transformer layers.</p>
          </div>

          <!-- Architecture Visual Diagram -->
          <div class="content-card">
            <div class="card-head">
              <h3>๐Ÿ“ Full Transformer Attention Interception Diagram</h3>
            </div>

            <div class="diagram-container">
              <div class="diagram-block block-input">
                <div class="block-title">1. Input Text & Token Embeddings</div>
                <div class="block-desc">User prompt text is converted into high-dimensional semantic token embeddings.</div>
              </div>

              <div class="diagram-arrow">โ–ผ</div>

              <div class="diagram-block block-transformer">
                <div class="block-title">2. Transformer Hidden Layer Stack (Layers 00 to 23)</div>
                <div class="block-desc">Multi-Head Self Attention processes Queries, Keys, and Values across all 24 transformer layers.</div>

                <!-- Inner Interception Layer -->
                <div class="rif-interception-box">
                  <div class="interception-badge">โšก KALPANA RIF CACHE LAYER (Drop-in Replacement for DynamicCache)</div>
                  
                  <div class="interception-grid">
                    <div class="sub-block">
                      <strong>Standard Transformers:</strong>
                      <code>torch.cat([Previous_Cache, New_Key_Tokens], dim=-2)</code>
                      <span class="val-rose">โŒ Unbounded Linear Memory Growth</span>
                    </div>
                    <div class="sub-block">
                      <strong>Kalpana RIF Substrate:</strong>
                      <code>KalpanaCacheLayer(past_key_values)</code>
                      <span class="val-emerald">โœ… Constant 96 MB Memory Across All 24 Layers</span>
                    </div>
                  </div>
                </div>
              </div>

              <div class="diagram-arrow">โ–ผ</div>

              <div class="diagram-block block-sweep">
                <div class="block-title">3. Continuous Wave Reconstruction & Attention Synthesis</div>
                <div class="block-desc">
                  Continuous wave memory channels deterministically reconstruct Key and Value attention states with zero memory expansion.
                </div>
              </div>

              <div class="diagram-arrow">โ–ผ</div>

              <div class="diagram-block block-output">
                <div class="block-title">4. Autoregressive Output Token Generation</div>
                <div class="block-desc">Generates response tokens with instant recall and zero recomputation overhead.</div>
              </div>
            </div>
          </div>

          <!-- End-to-End System Architecture Flow Card -->
          <div class="content-card" style="margin-top: 1.5rem;">
            <div class="card-head">
              <h3>๐ŸŒ End-to-End System Architecture & Dataflow Diagram</h3>
            </div>
            <div style="background: #080c18; border: 1px solid var(--border); border-radius: 8px; padding: 1.5rem; text-align: center;">
              <img src="https://raw.githubusercontent.com/maduperera/Kalpana-EmbedToKV/main/assets/kalpana_architecture.png" alt="Kalpana System Architecture Flow" style="max-width: 100%; max-height: 620px; object-fit: contain; border-radius: 6px; box-shadow: 0 4px 20px rgba(0, 0, 0, 0.5); background: #ffffff; padding: 12px;">
              <div style="font-size: 0.85rem; color: var(--text-muted); margin-top: 1rem; line-height: 1.5;">
                Complete pipeline: <strong>User Prompt</strong> โž” <strong>Tokenizer</strong> โž” <strong>Transformer Hidden Stack (24 Layers)</strong> โž” <strong>KalpanaDynamicCache (O(1))</strong> โž” <strong>Softmax Attention</strong> โž” <strong>Decoded Output</strong>.
              </div>
            </div>
          </div>

        </div>
      </section>

      <!-- ============================================================ -->
      <!-- TAB 4: SWAGGER / REST API DOCUMENTATION -->
      <!-- ============================================================ -->
      <section class="tab-pane" id="tab-swagger">
        <div class="pane-inner">
          <div class="section-header" style="display: flex; justify-content: space-between; align-items: center; flex-wrap: wrap; gap: 1rem;">
            <div>
              <h2>๐Ÿ”Œ Developer OpenAPI / Swagger API Reference</h2>
              <p>Standard REST inference and telemetry endpoints powered by dedicated NVIDIA GPU.</p>
            </div>
            <a href="https://huggingface.co/spaces/MaduRox/Kalpana-API-GPU" target="_blank" class="btn-primary" style="text-decoration: none; width: auto; padding: 0.6rem 1.2rem; display: inline-flex; align-items: center; gap: 0.5rem;">
              <span>๐Ÿ“– Open Kalpanฤ GPU Space โ†—๏ธ</span>
            </a>
          </div>

          <!-- Base URL Banner -->
          <div style="background: rgba(0, 240, 255, 0.05); border: 1px solid var(--border-cyan); border-radius: 8px; padding: 0.8rem 1.2rem; margin-bottom: 1.5rem; display: flex; justify-content: space-between; align-items: center;">
            <div>
              <span style="color: var(--text-muted); font-size: 0.8rem;">HOSTED GPU ENDPOINT:</span>
              <span style="font-family: var(--font-mono); font-weight: 700; color: var(--cyan); margin-left: 0.5rem;">https://madurox-kalpana-api-gpu.hf.space</span>
            </div>
            <span class="status-tag tag-pass">ONLINE ยท NVIDIA GPU (T4 DEDICATED)</span>
          </div>

          <!-- Windows PowerShell Example -->
          <div class="swagger-endpoint open">
            <div class="endpoint-header" onclick="toggleSwagger(this)">
              <span class="http-method method-post">POWERSHELL</span>
              <span class="endpoint-path">Windows PowerShell (Single-Command)</span>
              <span class="endpoint-summary">1-Click Execution using native Invoke-RestMethod</span>
              <span class="expand-icon">โ–ผ</span>
            </div>
            <div class="endpoint-body">
              <pre class="code-block"><code>$res = Invoke-RestMethod -Uri "https://madurox-kalpana-api-gpu.hf.space/gradio_api/call/generate" -Method Post -ContentType "application/json" -Body '{"data": ["What is cricket?", 128, 0.7]}'
Invoke-RestMethod -Uri "https://madurox-kalpana-api-gpu.hf.space/gradio_api/call/generate/$($res.event_id)"</code></pre>
            </div>
          </div>

          <!-- Python Requests Example -->
          <div class="swagger-endpoint">
            <div class="endpoint-header" onclick="toggleSwagger(this)">
              <span class="http-method method-post">PYTHON</span>
              <span class="endpoint-path">Python (requests REST Stream)</span>
              <span class="endpoint-summary">Universal 2-step REST streaming client</span>
              <span class="expand-icon">โ–ผ</span>
            </div>
            <div class="endpoint-body">
              <pre class="code-block"><code>import requests

# Step 1: Submit prompt
post_res = requests.post(
    "https://madurox-kalpana-api-gpu.hf.space/gradio_api/call/generate",
    json={"data": ["What is cricket?", 128, 0.7]}
)
event_id = post_res.json()["event_id"]

# Step 2: Stream response
sse_res = requests.get(f"https://madurox-kalpana-api-gpu.hf.space/gradio_api/call/generate/{event_id}")
for line in sse_res.text.split("\n"):
    if line.startswith("data:"):
        print("Generated Output:", line[5:])</code></pre>
            </div>
          </div>

          <!-- JavaScript Example -->
          <div class="swagger-endpoint">
            <div class="endpoint-header" onclick="toggleSwagger(this)">
              <span class="http-method method-post">JS / WEB</span>
              <span class="endpoint-path">JavaScript (fetch SSE Stream)</span>
              <span class="endpoint-summary">Web and mobile app client integration</span>
              <span class="expand-icon">โ–ผ</span>
            </div>
            <div class="endpoint-body">
              <pre class="code-block"><code>// Step 1: POST prompt
const postRes = await fetch("https://madurox-kalpana-api-gpu.hf.space/gradio_api/call/generate", {
  method: "POST",
  headers: { "Content-Type": "application/json" },
  body: JSON.stringify({ data: ["What is quantum superposition?", 128, 0.7] })
});
const { event_id } = await postRes.json();

// Step 2: Stream answer
const sseRes = await fetch(`https://madurox-kalpana-api-gpu.hf.space/gradio_api/call/generate/${event_id}`);
const text = await sseRes.text();
console.log("Output:", text);</code></pre>
            </div>
          </div>

        </div>
      </section>

      <!-- ============================================================ -->
      <!-- TAB 5: UNIT ECONOMICS & INVESTOR METRICS -->
      <!-- ============================================================ -->
      <section class="tab-pane" id="tab-economics">
        <div class="pane-inner">
          <div class="section-header">
            <h2>๐Ÿ’ฐ Unit Economics: 800+ Concurrent 1M-Token Contexts on 1 GPU</h2>
            <p>How Kalpana eliminates the $432/user/month KV Cache "GPU Tax" down to $2.75/user/month.</p>
          </div>

          <div class="stats-banner">
            <div class="stat-box">
              <div class="stat-number val-good">$2.75</div>
              <div class="stat-label">Cost per User / Month (1M Context)</div>
            </div>
            <div class="stat-box">
              <div class="stat-number">76.8 GB</div>
              <div class="stat-label">VRAM for 800 ร— 1M-Token Sessions</div>
            </div>
            <div class="stat-box">
              <div class="stat-number val-rose">110.4 TB</div>
              <div class="stat-label">Traditional VRAM Needed for 800 Users</div>
            </div>
            <div class="stat-box">
              <div class="stat-number val-cyan">1,437ร—</div>
              <div class="stat-label">VRAM Density Multiplication</div>
            </div>
          </div>

          <div class="content-card" style="margin-top: 1.5rem;">
            <div class="card-head">
              <h3>๐Ÿ’ต Traditional KV-Cache Cost Wall vs. Kalpana RIF ($/user/month)</h3>
            </div>
            <table class="data-table">
              <thead>
                <tr>
                  <th>Context Length</th>
                  <th>Traditional Cloud API Cost / User / Mo</th>
                  <th>Kalpana RIF Substrate Cost / User / Mo</th>
                  <th>Monthly Savings</th>
                </tr>
              </thead>
              <tbody>
                <tr>
                  <td><strong>2,000 tokens</strong></td>
                  <td>$7.45 / user</td>
                  <td><strong class="val-good">$2.75 / user</strong></td>
                  <td>2.7ร— cheaper</td>
                </tr>
                <tr>
                  <td><strong>8,000 tokens</strong></td>
                  <td>$28.80 / user</td>
                  <td><strong class="val-good">$2.75 / user</strong></td>
                  <td>10.5ร— cheaper</td>
                </tr>
                <tr>
                  <td><strong>32,000 tokens</strong></td>
                  <td>$114.00 / user</td>
                  <td><strong class="val-good">$2.75 / user</strong></td>
                  <td>41.5ร— cheaper</td>
                </tr>
                <tr>
                  <td><strong>128,000 tokens</strong></td>
                  <td>$432.00 / user</td>
                  <td><strong class="val-good">$2.75 / user</strong></td>
                  <td>157ร— cheaper</td>
                </tr>
                <tr>
                  <td><strong>1,000,000 tokens</strong></td>
                  <td><span class="val-rose">โˆž (Impractical - $10,000+)</span></td>
                  <td><strong class="val-good">$2.75 / user</strong></td>
                  <td><strong>3,600ร— cheaper</strong></td>
                </tr>
              </tbody>
            </table>
          </div>

        </div>
      </section>

    </div>
  </div>

  <script type="module" src="./app.js"></script>
</body>
</html>