Spaces:
Build error
Build error
Upload folder using huggingface_hub
Browse filesThis view is limited to 50 files because it contains too many changes. See raw diff
- .gitattributes +1 -0
- .gradio/certificate.pem +31 -0
- .ipynb_checkpoints/PELS_Training_RTX3050-checkpoint.ipynb +0 -0
- .ipynb_checkpoints/Untitled-checkpoint.ipynb +6 -0
- .ipynb_checkpoints/app-checkpoint.py +501 -0
- .ipynb_checkpoints/pels_correlation-checkpoint.png +0 -0
- PELS_Final_Justification.xlsx +3 -0
- PELS_Training_RTX3050.ipynb +0 -0
- README.md +2 -8
- Untitled.ipynb +223 -0
- app.py +501 -0
- pels_correlation.png +0 -0
- pels_phi2_qlora/README.md +202 -0
- pels_phi2_qlora/adapter_config.json +33 -0
- pels_phi2_qlora/adapter_model.safetensors +3 -0
- pels_phi2_qlora/added_tokens.json +40 -0
- pels_phi2_qlora/checkpoint-400/README.md +202 -0
- pels_phi2_qlora/checkpoint-400/adapter_config.json +33 -0
- pels_phi2_qlora/checkpoint-400/adapter_model.safetensors +3 -0
- pels_phi2_qlora/checkpoint-400/added_tokens.json +40 -0
- pels_phi2_qlora/checkpoint-400/merges.txt +0 -0
- pels_phi2_qlora/checkpoint-400/optimizer.pt +3 -0
- pels_phi2_qlora/checkpoint-400/rng_state.pth +3 -0
- pels_phi2_qlora/checkpoint-400/scheduler.pt +3 -0
- pels_phi2_qlora/checkpoint-400/special_tokens_map.json +24 -0
- pels_phi2_qlora/checkpoint-400/tokenizer.json +0 -0
- pels_phi2_qlora/checkpoint-400/tokenizer_config.json +325 -0
- pels_phi2_qlora/checkpoint-400/trainer_state.json +165 -0
- pels_phi2_qlora/checkpoint-400/training_args.bin +3 -0
- pels_phi2_qlora/checkpoint-400/vocab.json +0 -0
- pels_phi2_qlora/checkpoint-500/README.md +202 -0
- pels_phi2_qlora/checkpoint-500/adapter_config.json +33 -0
- pels_phi2_qlora/checkpoint-500/adapter_model.safetensors +3 -0
- pels_phi2_qlora/checkpoint-500/added_tokens.json +40 -0
- pels_phi2_qlora/checkpoint-500/merges.txt +0 -0
- pels_phi2_qlora/checkpoint-500/optimizer.pt +3 -0
- pels_phi2_qlora/checkpoint-500/rng_state.pth +3 -0
- pels_phi2_qlora/checkpoint-500/scheduler.pt +3 -0
- pels_phi2_qlora/checkpoint-500/special_tokens_map.json +24 -0
- pels_phi2_qlora/checkpoint-500/tokenizer.json +0 -0
- pels_phi2_qlora/checkpoint-500/tokenizer_config.json +325 -0
- pels_phi2_qlora/checkpoint-500/trainer_state.json +201 -0
- pels_phi2_qlora/checkpoint-500/training_args.bin +3 -0
- pels_phi2_qlora/checkpoint-500/vocab.json +0 -0
- pels_phi2_qlora/merges.txt +0 -0
- pels_phi2_qlora/special_tokens_map.json +24 -0
- pels_phi2_qlora/tokenizer.json +0 -0
- pels_phi2_qlora/tokenizer_config.json +325 -0
- pels_phi2_qlora/training_args.bin +3 -0
- pels_phi2_qlora/vocab.json +0 -0
.gitattributes
CHANGED
|
@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
+
PELS_Final_Justification.xlsx filter=lfs diff=lfs merge=lfs -text
|
.gradio/certificate.pem
ADDED
|
@@ -0,0 +1,31 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
-----BEGIN CERTIFICATE-----
|
| 2 |
+
MIIFazCCA1OgAwIBAgIRAIIQz7DSQONZRGPgu2OCiwAwDQYJKoZIhvcNAQELBQAw
|
| 3 |
+
TzELMAkGA1UEBhMCVVMxKTAnBgNVBAoTIEludGVybmV0IFNlY3VyaXR5IFJlc2Vh
|
| 4 |
+
cmNoIEdyb3VwMRUwEwYDVQQDEwxJU1JHIFJvb3QgWDEwHhcNMTUwNjA0MTEwNDM4
|
| 5 |
+
WhcNMzUwNjA0MTEwNDM4WjBPMQswCQYDVQQGEwJVUzEpMCcGA1UEChMgSW50ZXJu
|
| 6 |
+
ZXQgU2VjdXJpdHkgUmVzZWFyY2ggR3JvdXAxFTATBgNVBAMTDElTUkcgUm9vdCBY
|
| 7 |
+
MTCCAiIwDQYJKoZIhvcNAQEBBQADggIPADCCAgoCggIBAK3oJHP0FDfzm54rVygc
|
| 8 |
+
h77ct984kIxuPOZXoHj3dcKi/vVqbvYATyjb3miGbESTtrFj/RQSa78f0uoxmyF+
|
| 9 |
+
0TM8ukj13Xnfs7j/EvEhmkvBioZxaUpmZmyPfjxwv60pIgbz5MDmgK7iS4+3mX6U
|
| 10 |
+
A5/TR5d8mUgjU+g4rk8Kb4Mu0UlXjIB0ttov0DiNewNwIRt18jA8+o+u3dpjq+sW
|
| 11 |
+
T8KOEUt+zwvo/7V3LvSye0rgTBIlDHCNAymg4VMk7BPZ7hm/ELNKjD+Jo2FR3qyH
|
| 12 |
+
B5T0Y3HsLuJvW5iB4YlcNHlsdu87kGJ55tukmi8mxdAQ4Q7e2RCOFvu396j3x+UC
|
| 13 |
+
B5iPNgiV5+I3lg02dZ77DnKxHZu8A/lJBdiB3QW0KtZB6awBdpUKD9jf1b0SHzUv
|
| 14 |
+
KBds0pjBqAlkd25HN7rOrFleaJ1/ctaJxQZBKT5ZPt0m9STJEadao0xAH0ahmbWn
|
| 15 |
+
OlFuhjuefXKnEgV4We0+UXgVCwOPjdAvBbI+e0ocS3MFEvzG6uBQE3xDk3SzynTn
|
| 16 |
+
jh8BCNAw1FtxNrQHusEwMFxIt4I7mKZ9YIqioymCzLq9gwQbooMDQaHWBfEbwrbw
|
| 17 |
+
qHyGO0aoSCqI3Haadr8faqU9GY/rOPNk3sgrDQoo//fb4hVC1CLQJ13hef4Y53CI
|
| 18 |
+
rU7m2Ys6xt0nUW7/vGT1M0NPAgMBAAGjQjBAMA4GA1UdDwEB/wQEAwIBBjAPBgNV
|
| 19 |
+
HRMBAf8EBTADAQH/MB0GA1UdDgQWBBR5tFnme7bl5AFzgAiIyBpY9umbbjANBgkq
|
| 20 |
+
hkiG9w0BAQsFAAOCAgEAVR9YqbyyqFDQDLHYGmkgJykIrGF1XIpu+ILlaS/V9lZL
|
| 21 |
+
ubhzEFnTIZd+50xx+7LSYK05qAvqFyFWhfFQDlnrzuBZ6brJFe+GnY+EgPbk6ZGQ
|
| 22 |
+
3BebYhtF8GaV0nxvwuo77x/Py9auJ/GpsMiu/X1+mvoiBOv/2X/qkSsisRcOj/KK
|
| 23 |
+
NFtY2PwByVS5uCbMiogziUwthDyC3+6WVwW6LLv3xLfHTjuCvjHIInNzktHCgKQ5
|
| 24 |
+
ORAzI4JMPJ+GslWYHb4phowim57iaztXOoJwTdwJx4nLCgdNbOhdjsnvzqvHu7Ur
|
| 25 |
+
TkXWStAmzOVyyghqpZXjFaH3pO3JLF+l+/+sKAIuvtd7u+Nxe5AW0wdeRlN8NwdC
|
| 26 |
+
jNPElpzVmbUq4JUagEiuTDkHzsxHpFKVK7q4+63SM1N95R1NbdWhscdCb+ZAJzVc
|
| 27 |
+
oyi3B43njTOQ5yOf+1CceWxG1bQVs5ZufpsMljq4Ui0/1lvh+wjChP4kqKOJ2qxq
|
| 28 |
+
4RgqsahDYVvTH9w7jXbyLeiNdd8XM2w9U/t7y0Ff/9yi0GE44Za4rF2LN9d11TPA
|
| 29 |
+
mRGunUHBcnWEvgJBQl9nJEiU0Zsnvgc/ubhPgXRR4Xq37Z0j4r7g1SgEEzwxA57d
|
| 30 |
+
emyPxgcYxn/eR44/KJ4EBs+lVDR3veyJm+kXQ99b21/+jh5Xos1AnX5iItreGCc=
|
| 31 |
+
-----END CERTIFICATE-----
|
.ipynb_checkpoints/PELS_Training_RTX3050-checkpoint.ipynb
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
.ipynb_checkpoints/Untitled-checkpoint.ipynb
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"cells": [],
|
| 3 |
+
"metadata": {},
|
| 4 |
+
"nbformat": 4,
|
| 5 |
+
"nbformat_minor": 5
|
| 6 |
+
}
|
.ipynb_checkpoints/app-checkpoint.py
ADDED
|
@@ -0,0 +1,501 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# ── Cell: PELS Gradio App (Scenario-Based) ────────────────────────────────────
|
| 2 |
+
|
| 3 |
+
!pip install gradio -q
|
| 4 |
+
|
| 5 |
+
import gradio as gr
|
| 6 |
+
import torch
|
| 7 |
+
import re
|
| 8 |
+
import random
|
| 9 |
+
|
| 10 |
+
model.eval()
|
| 11 |
+
torch.cuda.empty_cache()
|
| 12 |
+
|
| 13 |
+
RUBRIC = """C1 Foundations (15%): Task clarity, role setup, AI awareness.
|
| 14 |
+
C2 Design (20%): Prompt structure, patterns (few-shot, CoT, role+task+constraint).
|
| 15 |
+
C3 Output Spec (20%): Format, length, tone, structure constraints.
|
| 16 |
+
C4 Domain Application (20%): Domain vocabulary, contextual accuracy.
|
| 17 |
+
C5 Ethics (15%): No harmful/biased framing. Score=1 triggers automatic Final Score override to 1.0.
|
| 18 |
+
C6 Metacognition (10%): Self-awareness, iteration design, fallback handling."""
|
| 19 |
+
|
| 20 |
+
MAX_LENGTH = 512
|
| 21 |
+
MAX_NEW_TOKENS = 150
|
| 22 |
+
|
| 23 |
+
# ── Scenario bank ─────────────────────────────────────────────────────────────
|
| 24 |
+
# Each scenario has:
|
| 25 |
+
# "scenario" — the situation shown to the user
|
| 26 |
+
# "task" — what they are asked to do (no hints on HOW to prompt)
|
| 27 |
+
# The user must figure out how to write the prompt themselves.
|
| 28 |
+
|
| 29 |
+
SCENARIOS = {
|
| 30 |
+
"CAREER": [
|
| 31 |
+
{
|
| 32 |
+
"scenario": "Rohan is a mechanical engineer with 5 years of experience who wants to switch into data science. He has done one online Python course but has no real projects yet. He has an interview at a data analytics firm next month.",
|
| 33 |
+
"task": "Write an AI prompt that helps Rohan prepare for this career transition.",
|
| 34 |
+
},
|
| 35 |
+
{
|
| 36 |
+
"scenario": "Priya has been a school teacher for 8 years and wants to move into corporate L&D (Learning & Development). She has no corporate experience but has designed curriculum and trained 200+ students.",
|
| 37 |
+
"task": "Write an AI prompt that helps Priya position herself for an L&D role.",
|
| 38 |
+
},
|
| 39 |
+
{
|
| 40 |
+
"scenario": "Amir graduated 6 months ago with a BCA degree and has been applying for software developer roles but getting no callbacks. His resume lists his college projects but no internships.",
|
| 41 |
+
"task": "Write an AI prompt that helps Amir fix the problem and get more callbacks.",
|
| 42 |
+
},
|
| 43 |
+
{
|
| 44 |
+
"scenario": "Sneha is a marketing manager at a mid-size company. She wants to ask for a promotion to Director but has never negotiated salary or title before and doesn't know how to make the case.",
|
| 45 |
+
"task": "Write an AI prompt that helps Sneha prepare for this conversation with her manager.",
|
| 46 |
+
},
|
| 47 |
+
],
|
| 48 |
+
"EDUCATION": [
|
| 49 |
+
{
|
| 50 |
+
"scenario": "A Grade 9 teacher needs to explain the concept of compound interest to students who understand basic multiplication and percentages but have never studied finance or banking.",
|
| 51 |
+
"task": "Write an AI prompt that produces a teaching resource for this class.",
|
| 52 |
+
},
|
| 53 |
+
{
|
| 54 |
+
"scenario": "A university professor wants to check whether her 2nd-year engineering students have understood Newton's Laws of Motion — specifically common misconceptions students have about inertia.",
|
| 55 |
+
"task": "Write an AI prompt that generates an assessment to test this understanding.",
|
| 56 |
+
},
|
| 57 |
+
{
|
| 58 |
+
"scenario": "A homeschooling parent needs to teach their 10-year-old child about climate change in a way that is factually accurate but not scary or overwhelming, using everyday examples.",
|
| 59 |
+
"task": "Write an AI prompt that creates an age-appropriate lesson on this topic.",
|
| 60 |
+
},
|
| 61 |
+
{
|
| 62 |
+
"scenario": "A coding bootcamp instructor wants to introduce recursion to students who are comfortable with loops (for/while) but have never seen a function call itself.",
|
| 63 |
+
"task": "Write an AI prompt that creates a beginner-friendly explanation with an exercise.",
|
| 64 |
+
},
|
| 65 |
+
],
|
| 66 |
+
"TECHNOLOGY": [
|
| 67 |
+
{
|
| 68 |
+
"scenario": "A junior developer at a startup wrote a Python script that reads a CSV file and calculates monthly sales totals — but it crashes whenever a cell is empty or contains text instead of a number.",
|
| 69 |
+
"task": "Write an AI prompt that helps fix and improve this script.",
|
| 70 |
+
},
|
| 71 |
+
{
|
| 72 |
+
"scenario": "A non-technical product manager needs to explain to her team why their app is slow. The engineering team says it is a 'database N+1 query problem' but she doesn't understand what that means.",
|
| 73 |
+
"task": "Write an AI prompt that produces an explanation she can actually understand and relay to stakeholders.",
|
| 74 |
+
},
|
| 75 |
+
{
|
| 76 |
+
"scenario": "A small business owner wants to build a simple website contact form that stores submissions in a Google Sheet — they have no coding experience and a budget of zero.",
|
| 77 |
+
"task": "Write an AI prompt that gives them a practical, step-by-step solution.",
|
| 78 |
+
},
|
| 79 |
+
{
|
| 80 |
+
"scenario": "A data analyst has a pandas DataFrame with 500,000 rows. Her current code takes 4 minutes to run a groupby aggregation. Her manager wants results in under 30 seconds.",
|
| 81 |
+
"task": "Write an AI prompt that helps her optimise the code.",
|
| 82 |
+
},
|
| 83 |
+
],
|
| 84 |
+
"HEALTHCARE": [
|
| 85 |
+
{
|
| 86 |
+
"scenario": "A 45-year-old patient was just diagnosed with pre-diabetes. Their doctor told them to 'watch their diet and exercise more' but gave no specific guidance. The patient is confused about what to actually do.",
|
| 87 |
+
"task": "Write an AI prompt that produces practical, safe guidance for this patient.",
|
| 88 |
+
},
|
| 89 |
+
{
|
| 90 |
+
"scenario": "A nurse manager at a clinic needs to train new staff on the correct procedure for hand hygiene according to WHO guidelines — in a way that is quick to read and easy to remember during a busy shift.",
|
| 91 |
+
"task": "Write an AI prompt that creates this training material.",
|
| 92 |
+
},
|
| 93 |
+
{
|
| 94 |
+
"scenario": "A medical student is struggling to remember the differences between Type 1 and Type 2 Diabetes — the symptoms, causes, treatment approaches, and which patient populations are typically affected.",
|
| 95 |
+
"task": "Write an AI prompt that creates a study aid for this topic.",
|
| 96 |
+
},
|
| 97 |
+
{
|
| 98 |
+
"scenario": "A hospital wants to send a clear, non-alarming message to patients reminding them to get their annual flu vaccination — the message will go out via SMS so it must be very short.",
|
| 99 |
+
"task": "Write an AI prompt that generates this patient communication.",
|
| 100 |
+
},
|
| 101 |
+
],
|
| 102 |
+
"LEGAL": [
|
| 103 |
+
{
|
| 104 |
+
"scenario": "A freelance graphic designer in Pune completed a logo project for a client who is now refusing to pay the ₹25,000 invoice, claiming the work was 'not what was agreed'. There was a WhatsApp conversation but no formal contract.",
|
| 105 |
+
"task": "Write an AI prompt that helps the designer understand their options and next steps.",
|
| 106 |
+
},
|
| 107 |
+
{
|
| 108 |
+
"scenario": "A first-time landlord in Bangalore wants to rent out their apartment. They have heard that verbal agreements can cause problems and want to create a proper rental agreement that protects them legally.",
|
| 109 |
+
"task": "Write an AI prompt that helps them draft or understand what should be in this agreement.",
|
| 110 |
+
},
|
| 111 |
+
{
|
| 112 |
+
"scenario": "An employee received a termination letter from their company citing 'performance issues' but believes they were fired because they filed a complaint against their manager last month.",
|
| 113 |
+
"task": "Write an AI prompt that helps this person understand whether they have a case and what to do.",
|
| 114 |
+
},
|
| 115 |
+
{
|
| 116 |
+
"scenario": "A startup founder is about to sign a 3-year office lease. They have never signed a commercial lease before and are worried about clauses that could trap them if the startup fails in year one.",
|
| 117 |
+
"task": "Write an AI prompt that helps them know what to watch out for in this agreement.",
|
| 118 |
+
},
|
| 119 |
+
],
|
| 120 |
+
"MARKETING": [
|
| 121 |
+
{
|
| 122 |
+
"scenario": "A local bakery in Hyderabad has great reviews but almost no online presence. They want to attract customers aged 20–35 who discover food businesses on Instagram. Their budget is zero — only organic content.",
|
| 123 |
+
"task": "Write an AI prompt that helps them create a content strategy or specific post.",
|
| 124 |
+
},
|
| 125 |
+
{
|
| 126 |
+
"scenario": "An edtech startup is launching a new course on AI for non-technical professionals. They need to send a launch email to their existing subscriber list of 5,000 people — most of whom haven't opened emails in 3 months.",
|
| 127 |
+
"task": "Write an AI prompt that produces this re-engagement launch email.",
|
| 128 |
+
},
|
| 129 |
+
{
|
| 130 |
+
"scenario": "A fitness trainer wants to run Google Ads for their personal training services in Mumbai. They have a ₹10,000/month budget and have never run paid ads before. Their USP is online coaching with personalised meal plans.",
|
| 131 |
+
"task": "Write an AI prompt that helps them set up or plan this campaign effectively.",
|
| 132 |
+
},
|
| 133 |
+
{
|
| 134 |
+
"scenario": "A sustainable clothing brand is launching a new line made from recycled ocean plastic. Their target customer cares about the environment but is price-sensitive (products are 30% more expensive than fast fashion).",
|
| 135 |
+
"task": "Write an AI prompt that creates compelling product description copy for their website.",
|
| 136 |
+
},
|
| 137 |
+
],
|
| 138 |
+
"FINANCE": [
|
| 139 |
+
{
|
| 140 |
+
"scenario": "A 28-year-old software developer earns ₹1.2 lakh per month but saves almost nothing. They have ₹3 lakh in credit card debt at 36% annual interest and no investments. They want to start getting their finances in order.",
|
| 141 |
+
"task": "Write an AI prompt that produces a practical financial plan for this person.",
|
| 142 |
+
},
|
| 143 |
+
{
|
| 144 |
+
"scenario": "A small business owner wants to understand their company's cash flow statement. Their accountant gave them a document but they don't understand why the business is profitable on paper but always short on cash.",
|
| 145 |
+
"task": "Write an AI prompt that explains this concept in a way they can immediately apply to their situation.",
|
| 146 |
+
},
|
| 147 |
+
{
|
| 148 |
+
"scenario": "A couple wants to save for their child's higher education. The child is currently 3 years old and they estimate they'll need ₹30 lakhs in 15 years. They can invest ₹10,000 per month.",
|
| 149 |
+
"task": "Write an AI prompt that helps them understand their investment options and whether their goal is achievable.",
|
| 150 |
+
},
|
| 151 |
+
{
|
| 152 |
+
"scenario": "A salaried employee received their first Form 16 and has to file their ITR for the first time. They are confused about which ITR form to use, what deductions they can claim under 80C, and how to avoid mistakes.",
|
| 153 |
+
"task": "Write an AI prompt that guides them through this process.",
|
| 154 |
+
},
|
| 155 |
+
],
|
| 156 |
+
"CREATIVE": [
|
| 157 |
+
{
|
| 158 |
+
"scenario": "A screenwriter wants to write a 5-minute short film about loneliness in a big city. The protagonist is a 30-year-old delivery driver who interacts with dozens of people every day but has no real relationships.",
|
| 159 |
+
"task": "Write an AI prompt that helps develop this concept into a concrete scene or script outline.",
|
| 160 |
+
},
|
| 161 |
+
{
|
| 162 |
+
"scenario": "A startup founder needs to write the 'About Us' page for their company website. The company builds AI tools for teachers. The tone should feel human and mission-driven, not corporate or salesy.",
|
| 163 |
+
"task": "Write an AI prompt that produces this About Us page.",
|
| 164 |
+
},
|
| 165 |
+
{
|
| 166 |
+
"scenario": "A children's book author wants to write a short story for 6–8 year olds that teaches them about the importance of asking for help — without being preachy. The story should have an animal character.",
|
| 167 |
+
"task": "Write an AI prompt that generates this story or a strong outline for it.",
|
| 168 |
+
},
|
| 169 |
+
{
|
| 170 |
+
"scenario": "A musician wants to write lyrics for an indie-folk song about their grandmother who passed away last year. The mood should be bittersweet — celebrating her life rather than mourning — with imagery from her kitchen and garden.",
|
| 171 |
+
"task": "Write an AI prompt that helps generate these lyrics or a draft verse.",
|
| 172 |
+
},
|
| 173 |
+
],
|
| 174 |
+
}
|
| 175 |
+
|
| 176 |
+
# ── Core functions ────────────────────────────────────────────────────────────
|
| 177 |
+
def get_scenario(domain):
|
| 178 |
+
if not domain:
|
| 179 |
+
return "", ""
|
| 180 |
+
scenarios = SCENARIOS.get(domain, [])
|
| 181 |
+
if not scenarios:
|
| 182 |
+
return "", ""
|
| 183 |
+
picked = random.choice(scenarios)
|
| 184 |
+
scenario_html = f"""
|
| 185 |
+
<div style="background:#0f172a;border-radius:10px;padding:20px 22px;
|
| 186 |
+
border-left:4px solid #6366f1;font-family:sans-serif">
|
| 187 |
+
<div style="color:#a5b4fc;font-size:11px;font-weight:700;
|
| 188 |
+
letter-spacing:.1em;margin-bottom:10px">
|
| 189 |
+
📋 YOUR SCENARIO
|
| 190 |
+
</div>
|
| 191 |
+
<div style="color:#e2e8f0;font-size:15px;line-height:1.75;margin-bottom:16px">
|
| 192 |
+
{picked['scenario']}
|
| 193 |
+
</div>
|
| 194 |
+
<div style="background:#1e293b;border-radius:8px;padding:12px 16px">
|
| 195 |
+
<span style="color:#fbbf24;font-weight:700;font-size:13px">🎯 Your Task: </span>
|
| 196 |
+
<span style="color:#fde68a;font-size:14px">{picked['task']}</span>
|
| 197 |
+
</div>
|
| 198 |
+
<div style="margin-top:14px;color:#475569;font-size:12px;font-style:italic">
|
| 199 |
+
💡 Tip: A strong prompt assigns a role to the AI, specifies the audience,
|
| 200 |
+
defines the output format, and sets constraints. You figure out how — that's the assessment.
|
| 201 |
+
</div>
|
| 202 |
+
</div>"""
|
| 203 |
+
return scenario_html, picked["scenario"] + " | Task: " + picked["task"]
|
| 204 |
+
|
| 205 |
+
def get_tier(score):
|
| 206 |
+
"""score is always 0–10"""
|
| 207 |
+
if score >= 7.1: return "🟢 Strong", "#22c55e"
|
| 208 |
+
elif score >= 4.1: return "🟡 Developing", "#f59e0b"
|
| 209 |
+
else: return "🔴 Weak", "#ef4444"
|
| 210 |
+
|
| 211 |
+
def extract_scores(text):
|
| 212 |
+
patterns = {
|
| 213 |
+
'c1': r'C1_Foundations:\s*([0-9.]+)',
|
| 214 |
+
'c2': r'C2_Design:\s*([0-9.]+)',
|
| 215 |
+
'c3': r'C3_OutputSpec:\s*([0-9.]+)',
|
| 216 |
+
'c4': r'C4_Domain:\s*([0-9.]+)',
|
| 217 |
+
'c5': r'C5_Ethics:\s*([0-9.]+)',
|
| 218 |
+
'c6': r'C6_Metacognition:\s*([0-9.]+)',
|
| 219 |
+
'final': r'Final_Score:\s*([0-9.]+)',
|
| 220 |
+
}
|
| 221 |
+
return {k: float(m.group(1)) if (m := re.search(p, text)) else None
|
| 222 |
+
for k, p in patterns.items()}
|
| 223 |
+
|
| 224 |
+
def grade_prompt_fast(prompt_text, domain):
|
| 225 |
+
input_text = (
|
| 226 |
+
f"Instruct: ### PELS Grading Task\n"
|
| 227 |
+
f"Domain: {domain}\n"
|
| 228 |
+
f"Rubric:\n{RUBRIC}\n\n"
|
| 229 |
+
f"Candidate Prompt:\n{prompt_text}\n\n"
|
| 230 |
+
f"### Evaluation\n"
|
| 231 |
+
f"Score each category 1–10. Ethics score of 1 overrides all others.\n"
|
| 232 |
+
f"Output:"
|
| 233 |
+
)
|
| 234 |
+
inputs = tokenizer(
|
| 235 |
+
input_text, return_tensors='pt',
|
| 236 |
+
truncation=True, max_length=MAX_LENGTH
|
| 237 |
+
).to(model.device)
|
| 238 |
+
|
| 239 |
+
with torch.no_grad():
|
| 240 |
+
outputs = model.generate(
|
| 241 |
+
**inputs,
|
| 242 |
+
max_new_tokens=MAX_NEW_TOKENS,
|
| 243 |
+
do_sample=False,
|
| 244 |
+
pad_token_id=tokenizer.eos_token_id,
|
| 245 |
+
)
|
| 246 |
+
decoded = tokenizer.decode(outputs[0], skip_special_tokens=True)
|
| 247 |
+
out_part = decoded[len(input_text):] if input_text in decoded else decoded
|
| 248 |
+
return out_part, extract_scores(out_part)
|
| 249 |
+
|
| 250 |
+
def build_score_html(scores, veto, raw_output):
|
| 251 |
+
final = scores.get("final") or 0.0
|
| 252 |
+
final_display = final if final > 1.0 else final * 10
|
| 253 |
+
bar_pct = min(int(final_display * 10), 100)
|
| 254 |
+
t_label, color = get_tier(final_display)
|
| 255 |
+
|
| 256 |
+
cat_info = [
|
| 257 |
+
("c1", "C1 · Foundations", "15%"),
|
| 258 |
+
("c2", "C2 · Design", "20%"),
|
| 259 |
+
("c3", "C3 · Output Spec", "20%"),
|
| 260 |
+
("c4", "C4 · Domain Application", "20%"),
|
| 261 |
+
("c5", "C5 · Ethics", "15%"),
|
| 262 |
+
("c6", "C6 · Metacognition", "10%"),
|
| 263 |
+
]
|
| 264 |
+
|
| 265 |
+
rows = ""
|
| 266 |
+
for key, label, weight in cat_info:
|
| 267 |
+
val = scores.get(key)
|
| 268 |
+
if val is None:
|
| 269 |
+
rows += f"""<tr>
|
| 270 |
+
<td style="padding:8px 12px;color:#94a3b8;font-family:sans-serif;
|
| 271 |
+
white-space:nowrap">{label} <span style="color:#334155">({weight})</span></td>
|
| 272 |
+
<td colspan="2" style="color:#475569;padding:8px 12px">—</td>
|
| 273 |
+
</tr>"""
|
| 274 |
+
continue
|
| 275 |
+
val_d = val if val > 1.0 else val * 10
|
| 276 |
+
_, c = get_tier(val_d)
|
| 277 |
+
pct = min(int(val_d * 10), 100)
|
| 278 |
+
rows += f"""<tr>
|
| 279 |
+
<td style="padding:8px 12px;font-weight:600;color:#cbd5e1;
|
| 280 |
+
white-space:nowrap;font-family:sans-serif">
|
| 281 |
+
{label} <span style="color:#475569;font-weight:400">({weight})</span>
|
| 282 |
+
</td>
|
| 283 |
+
<td style="padding:8px 12px;width:100%">
|
| 284 |
+
<div style="background:#1e293b;border-radius:4px;height:13px;overflow:hidden">
|
| 285 |
+
<div style="background:{c};width:{pct}%;height:100%;border-radius:4px"></div>
|
| 286 |
+
</div>
|
| 287 |
+
</td>
|
| 288 |
+
<td style="padding:8px 12px;font-weight:700;color:{c};
|
| 289 |
+
white-space:nowrap;font-family:sans-serif">{val_d:.1f}/10</td>
|
| 290 |
+
</tr>"""
|
| 291 |
+
|
| 292 |
+
veto_banner = ""
|
| 293 |
+
if veto:
|
| 294 |
+
veto_banner = """<div style="background:#7f1d1d;border:1px solid #dc2626;
|
| 295 |
+
border-radius:8px;padding:10px 16px;margin-bottom:16px;
|
| 296 |
+
color:#fca5a5;font-weight:600;font-family:sans-serif">
|
| 297 |
+
⛔ Ethics Veto Triggered — Final Score overridden to 1.0
|
| 298 |
+
</div>"""
|
| 299 |
+
|
| 300 |
+
just_match = re.search(r'Justification:\s*(.+)', raw_output, re.DOTALL)
|
| 301 |
+
justification = just_match.group(1).strip()[:500] if just_match else ""
|
| 302 |
+
just_html = ""
|
| 303 |
+
if justification:
|
| 304 |
+
just_html = f"""
|
| 305 |
+
<div style="margin-top:20px;padding:14px 16px;background:#0f172a;
|
| 306 |
+
border-left:3px solid #6366f1;border-radius:0 8px 8px 0">
|
| 307 |
+
<div style="color:#a5b4fc;font-size:11px;font-weight:700;
|
| 308 |
+
letter-spacing:.08em;margin-bottom:8px;font-family:sans-serif">
|
| 309 |
+
JUSTIFICATION
|
| 310 |
+
</div>
|
| 311 |
+
<div style="color:#cbd5e1;line-height:1.7;font-size:14px;font-family:sans-serif">
|
| 312 |
+
{justification}
|
| 313 |
+
</div>
|
| 314 |
+
</div>"""
|
| 315 |
+
|
| 316 |
+
return f"""
|
| 317 |
+
<div style="font-family:sans-serif;background:#0f172a;border-radius:12px;
|
| 318 |
+
padding:24px;color:#e2e8f0">
|
| 319 |
+
{veto_banner}
|
| 320 |
+
<div style="display:flex;align-items:center;gap:24px;margin-bottom:24px">
|
| 321 |
+
<div style="text-align:center;min-width:90px">
|
| 322 |
+
<div style="font-size:54px;font-weight:800;color:{color};line-height:1">
|
| 323 |
+
{final_display:.1f}
|
| 324 |
+
</div>
|
| 325 |
+
<div style="font-size:12px;color:#64748b;margin-top:4px">out of 10</div>
|
| 326 |
+
</div>
|
| 327 |
+
<div style="flex:1">
|
| 328 |
+
<div style="background:#1e293b;border-radius:8px;height:20px;overflow:hidden">
|
| 329 |
+
<div style="background:{color};width:{bar_pct}%;height:100%;border-radius:8px">
|
| 330 |
+
</div>
|
| 331 |
+
</div>
|
| 332 |
+
<div style="margin-top:10px;font-size:17px;font-weight:700;color:{color}">
|
| 333 |
+
{t_label}
|
| 334 |
+
</div>
|
| 335 |
+
<div style="font-size:12px;color:#475569;margin-top:3px">PELS Final Score</div>
|
| 336 |
+
</div>
|
| 337 |
+
</div>
|
| 338 |
+
<table style="width:100%;border-collapse:collapse;font-size:14px">{rows}</table>
|
| 339 |
+
{just_html}
|
| 340 |
+
</div>"""
|
| 341 |
+
|
| 342 |
+
# ── Handlers ──────────────────────────────────────────────────────────────────
|
| 343 |
+
def on_domain_change(domain):
|
| 344 |
+
html, context = get_scenario(domain)
|
| 345 |
+
return html, context, "", PLACEHOLDER
|
| 346 |
+
|
| 347 |
+
def on_new_scenario(domain):
|
| 348 |
+
html, context = get_scenario(domain)
|
| 349 |
+
return html, context, "", PLACEHOLDER
|
| 350 |
+
|
| 351 |
+
def evaluate(domain, user_prompt, scenario_context):
|
| 352 |
+
if not domain:
|
| 353 |
+
return "<p style='color:#f87171;font-family:sans-serif'>⚠️ Please select a domain first.</p>"
|
| 354 |
+
if not user_prompt or len(user_prompt.strip()) < 10:
|
| 355 |
+
return "<p style='color:#f87171;font-family:sans-serif'>⚠️ Please write your prompt (at least 10 characters).</p>"
|
| 356 |
+
try:
|
| 357 |
+
raw, scores = grade_prompt_fast(user_prompt.strip(), domain)
|
| 358 |
+
except Exception as e:
|
| 359 |
+
return f"<p style='color:#f87171;font-family:sans-serif'>❌ Error: {e}</p>"
|
| 360 |
+
|
| 361 |
+
if not scores or all(v is None for v in scores.values()):
|
| 362 |
+
return f"""<div style='padding:16px;background:#1e293b;border-radius:8px;
|
| 363 |
+
color:#cbd5e1;font-family:sans-serif'>
|
| 364 |
+
<b>Raw model output:</b><br>
|
| 365 |
+
<pre style='white-space:pre-wrap;font-size:12px;color:#94a3b8'>{raw[:600]}</pre>
|
| 366 |
+
<small style='color:#64748b'>Scores not parsed. Try a more detailed prompt.</small>
|
| 367 |
+
</div>"""
|
| 368 |
+
|
| 369 |
+
veto = (scores.get("c5") or 10) <= 1.0
|
| 370 |
+
if veto:
|
| 371 |
+
scores["final"] = 1.0
|
| 372 |
+
|
| 373 |
+
return build_score_html(scores, veto, raw)
|
| 374 |
+
|
| 375 |
+
def clear_all():
|
| 376 |
+
return None, "", "", PLACEHOLDER
|
| 377 |
+
|
| 378 |
+
PLACEHOLDER = """
|
| 379 |
+
<div style='background:#0f172a;border-radius:12px;padding:40px;
|
| 380 |
+
text-align:center;color:#334155;font-family:sans-serif;font-size:15px'>
|
| 381 |
+
Your PELS score report will appear here after evaluation.
|
| 382 |
+
</div>"""
|
| 383 |
+
|
| 384 |
+
# ── UI ────────────────────────────────────────────────────────────────────────
|
| 385 |
+
with gr.Blocks(title="PELS Prompt Assessment", theme=gr.themes.Base()) as demo:
|
| 386 |
+
|
| 387 |
+
scenario_context = gr.State("") # hidden state stores scenario text
|
| 388 |
+
|
| 389 |
+
gr.HTML("""
|
| 390 |
+
<div style="padding:24px 0 12px">
|
| 391 |
+
<h1 style="font-family:sans-serif;font-size:2rem;font-weight:800;margin:0;
|
| 392 |
+
background:linear-gradient(135deg,#6366f1,#a855f7,#ec4899);
|
| 393 |
+
-webkit-background-clip:text;-webkit-text-fill-color:transparent">
|
| 394 |
+
PELS Prompt Assessment
|
| 395 |
+
</h1>
|
| 396 |
+
<p style="font-family:sans-serif;color:#64748b;margin:6px 0 0;font-size:14px">
|
| 397 |
+
You will receive a real-world scenario. Write an AI prompt to address it.
|
| 398 |
+
Your prompt will be evaluated across 6 rubric categories.
|
| 399 |
+
</p>
|
| 400 |
+
</div>""")
|
| 401 |
+
|
| 402 |
+
with gr.Row():
|
| 403 |
+
|
| 404 |
+
# ── Left ──────────────────────────────────────────────────────────────
|
| 405 |
+
with gr.Column(scale=1):
|
| 406 |
+
|
| 407 |
+
gr.HTML('<div style="font-family:sans-serif;font-weight:700;'
|
| 408 |
+
'color:#a5b4fc;font-size:12px;letter-spacing:.08em;'
|
| 409 |
+
'margin-bottom:6px">STEP 1 · CHOOSE DOMAIN</div>')
|
| 410 |
+
|
| 411 |
+
domain_dd = gr.Dropdown(
|
| 412 |
+
choices=list(SCENARIOS.keys()),
|
| 413 |
+
label="Domain", value=None, interactive=True,
|
| 414 |
+
)
|
| 415 |
+
|
| 416 |
+
gr.HTML('<div style="font-family:sans-serif;font-weight:700;'
|
| 417 |
+
'color:#a5b4fc;font-size:12px;letter-spacing:.08em;'
|
| 418 |
+
'margin:18px 0 6px">STEP 2 · YOUR SCENARIO</div>')
|
| 419 |
+
|
| 420 |
+
scenario_display = gr.HTML(
|
| 421 |
+
value="""<div style='background:#0f172a;border-radius:10px;
|
| 422 |
+
padding:20px;text-align:center;color:#334155;
|
| 423 |
+
font-family:sans-serif'>
|
| 424 |
+
Select a domain above to receive your scenario.
|
| 425 |
+
</div>"""
|
| 426 |
+
)
|
| 427 |
+
|
| 428 |
+
new_scenario_btn = gr.Button(
|
| 429 |
+
"🔀 Get Different Scenario", variant="secondary", size="sm"
|
| 430 |
+
)
|
| 431 |
+
|
| 432 |
+
gr.HTML('<div style="font-family:sans-serif;font-weight:700;'
|
| 433 |
+
'color:#a5b4fc;font-size:12px;letter-spacing:.08em;'
|
| 434 |
+
'margin:18px 0 6px">STEP 3 · WRITE YOUR PROMPT</div>')
|
| 435 |
+
|
| 436 |
+
prompt_box = gr.Textbox(
|
| 437 |
+
label="Your AI Prompt",
|
| 438 |
+
placeholder="Based on the scenario above, write your AI prompt here…",
|
| 439 |
+
lines=9,
|
| 440 |
+
)
|
| 441 |
+
|
| 442 |
+
with gr.Row():
|
| 443 |
+
clear_btn = gr.Button("🗑 Clear", variant="secondary", size="sm")
|
| 444 |
+
eval_btn = gr.Button("⚡ Evaluate", variant="primary", size="lg")
|
| 445 |
+
|
| 446 |
+
gr.HTML("""
|
| 447 |
+
<div style="margin-top:14px;padding:12px 14px;background:#1e293b;
|
| 448 |
+
border-radius:8px;font-size:12px;color:#64748b;
|
| 449 |
+
font-family:sans-serif;line-height:1.8">
|
| 450 |
+
<b style="color:#94a3b8">Rubric weights:</b><br>
|
| 451 |
+
C1 Foundations 15% · C2 Design 20% · C3 Output Spec 20%<br>
|
| 452 |
+
C4 Domain 20% · C5 Ethics 15% · C6 Metacognition 10%<br><br>
|
| 453 |
+
<b style="color:#94a3b8">Tiers:</b>
|
| 454 |
+
🟢 Strong ≥ 7.1
|
| 455 |
+
· 🟡 Developing 4.1–7.0
|
| 456 |
+
· 🔴 Weak ≤ 4.0
|
| 457 |
+
</div>""")
|
| 458 |
+
|
| 459 |
+
# ── Right ────��────────────────────────────────────────────────────────
|
| 460 |
+
with gr.Column(scale=1):
|
| 461 |
+
|
| 462 |
+
gr.HTML('<div style="font-family:sans-serif;font-weight:700;'
|
| 463 |
+
'color:#a5b4fc;font-size:12px;letter-spacing:.08em;'
|
| 464 |
+
'margin-bottom:6px">STEP 4 · PELS SCORE REPORT</div>')
|
| 465 |
+
|
| 466 |
+
result_html = gr.HTML(value=PLACEHOLDER)
|
| 467 |
+
|
| 468 |
+
with gr.Accordion("📖 Rubric Reference", open=False):
|
| 469 |
+
gr.Markdown("""
|
| 470 |
+
| Category | Weight | What it checks |
|
| 471 |
+
|---|---|---|
|
| 472 |
+
| C1 Foundations | 15% | Task clarity, role setup, AI awareness |
|
| 473 |
+
| C2 Design | 20% | Prompt structure — few-shot, CoT, role+task+constraint patterns |
|
| 474 |
+
| C3 Output Spec | 20% | Format, length, tone, structure constraints |
|
| 475 |
+
| C4 Domain | 20% | Domain vocabulary and contextual accuracy |
|
| 476 |
+
| C5 Ethics | 15% | No harmful or biased framing — score of 1 overrides the final score |
|
| 477 |
+
| C6 Metacognition | 10% | Self-awareness, iteration design, fallback handling |
|
| 478 |
+
""")
|
| 479 |
+
|
| 480 |
+
# ── Events ────────────────────────────────────────────────────────────────
|
| 481 |
+
domain_dd.change(
|
| 482 |
+
fn=on_domain_change,
|
| 483 |
+
inputs=domain_dd,
|
| 484 |
+
outputs=[scenario_display, scenario_context, prompt_box, result_html]
|
| 485 |
+
)
|
| 486 |
+
new_scenario_btn.click(
|
| 487 |
+
fn=on_new_scenario,
|
| 488 |
+
inputs=domain_dd,
|
| 489 |
+
outputs=[scenario_display, scenario_context, prompt_box, result_html]
|
| 490 |
+
)
|
| 491 |
+
eval_btn.click(
|
| 492 |
+
fn=evaluate,
|
| 493 |
+
inputs=[domain_dd, prompt_box, scenario_context],
|
| 494 |
+
outputs=result_html
|
| 495 |
+
)
|
| 496 |
+
clear_btn.click(
|
| 497 |
+
fn=clear_all,
|
| 498 |
+
outputs=[domain_dd, scenario_display, prompt_box, result_html]
|
| 499 |
+
)
|
| 500 |
+
|
| 501 |
+
demo.launch(share=True)
|
.ipynb_checkpoints/pels_correlation-checkpoint.png
ADDED
|
PELS_Final_Justification.xlsx
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:de28ad9db7b9b90b950a2645c27ae67e3fd38aa1baf78c912473cd7cdeb5cc27
|
| 3 |
+
size 496364
|
PELS_Training_RTX3050.ipynb
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
README.md
CHANGED
|
@@ -1,12 +1,6 @@
|
|
| 1 |
---
|
| 2 |
-
title:
|
| 3 |
-
|
| 4 |
-
colorFrom: green
|
| 5 |
-
colorTo: purple
|
| 6 |
sdk: gradio
|
| 7 |
sdk_version: 6.13.0
|
| 8 |
-
app_file: app.py
|
| 9 |
-
pinned: false
|
| 10 |
---
|
| 11 |
-
|
| 12 |
-
Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference
|
|
|
|
| 1 |
---
|
| 2 |
+
title: PELS_Grader
|
| 3 |
+
app_file: app.py
|
|
|
|
|
|
|
| 4 |
sdk: gradio
|
| 5 |
sdk_version: 6.13.0
|
|
|
|
|
|
|
| 6 |
---
|
|
|
|
|
|
Untitled.ipynb
ADDED
|
@@ -0,0 +1,223 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"cells": [
|
| 3 |
+
{
|
| 4 |
+
"cell_type": "code",
|
| 5 |
+
"execution_count": 1,
|
| 6 |
+
"id": "e3dbf502-0766-495f-93cd-672148b8ab6c",
|
| 7 |
+
"metadata": {},
|
| 8 |
+
"outputs": [
|
| 9 |
+
{
|
| 10 |
+
"name": "stdout",
|
| 11 |
+
"output_type": "stream",
|
| 12 |
+
"text": [
|
| 13 |
+
"✓ bitsandbytes 0.49.2 is already installed.\n"
|
| 14 |
+
]
|
| 15 |
+
}
|
| 16 |
+
],
|
| 17 |
+
"source": [
|
| 18 |
+
"# @title STEP 1 — Robust Dependency Installation\n",
|
| 19 |
+
"import subprocess, sys, importlib.metadata\n",
|
| 20 |
+
"\n",
|
| 21 |
+
"def install_packages():\n",
|
| 22 |
+
" print(\"Installing dependencies... this may take a minute.\")\n",
|
| 23 |
+
" packages = [\n",
|
| 24 |
+
" \"bitsandbytes>=0.45.3\",\n",
|
| 25 |
+
" \"transformers==4.46.3\",\n",
|
| 26 |
+
" \"peft==0.14.0\",\n",
|
| 27 |
+
" \"trl==0.12.2\",\n",
|
| 28 |
+
" \"accelerate==0.34.2\",\n",
|
| 29 |
+
" \"datasets==3.0.1\",\n",
|
| 30 |
+
" \"openpyxl\", \"scikit-learn\", \"scipy\", \"einops\", \"torchvision\"\n",
|
| 31 |
+
" ]\n",
|
| 32 |
+
" # Clean up potentially broken installations\n",
|
| 33 |
+
" subprocess.run([sys.executable, \"-m\", \"pip\", \"uninstall\", \"-y\", \"triton\", \"bitsandbytes\"], capture_output=True)\n",
|
| 34 |
+
" # Install fresh with verified versions\n",
|
| 35 |
+
" result = subprocess.run([sys.executable, \"-m\", \"pip\", \"install\", \"-q\", \"--upgrade\"] + packages)\n",
|
| 36 |
+
" if result.returncode == 0:\n",
|
| 37 |
+
" print(\"\\n✓ All dependencies installed successfully.\")\n",
|
| 38 |
+
" else:\n",
|
| 39 |
+
" print(\"\\n⚠ Installation encountered issues. Please check your internet connection or runtime.\")\n",
|
| 40 |
+
"\n",
|
| 41 |
+
"try:\n",
|
| 42 |
+
" # Check if bitsandbytes is fully registered with metadata\n",
|
| 43 |
+
" version = importlib.metadata.version(\"bitsandbytes\")\n",
|
| 44 |
+
" print(f\"✓ bitsandbytes {version} is already installed.\")\n",
|
| 45 |
+
"except (ImportError, importlib.metadata.PackageNotFoundError):\n",
|
| 46 |
+
" install_packages()\n",
|
| 47 |
+
" print(\"\\nCRITICAL: Please click 'Restart session' if prompted by Colab, then run from STEP 2.\")"
|
| 48 |
+
]
|
| 49 |
+
},
|
| 50 |
+
{
|
| 51 |
+
"cell_type": "code",
|
| 52 |
+
"execution_count": 2,
|
| 53 |
+
"id": "9f745318-0e12-4da8-9f9b-9e64e1511eb8",
|
| 54 |
+
"metadata": {},
|
| 55 |
+
"outputs": [
|
| 56 |
+
{
|
| 57 |
+
"name": "stdout",
|
| 58 |
+
"output_type": "stream",
|
| 59 |
+
"text": [
|
| 60 |
+
"Applying bitsandbytes environment fix...\n",
|
| 61 |
+
"✓ bnb version: 0.49.2\n",
|
| 62 |
+
"✓ Transformers-BNB integration check passed.\n"
|
| 63 |
+
]
|
| 64 |
+
}
|
| 65 |
+
],
|
| 66 |
+
"source": [
|
| 67 |
+
"import os\n",
|
| 68 |
+
"import sys\n",
|
| 69 |
+
"import subprocess\n",
|
| 70 |
+
"\n",
|
| 71 |
+
"# Force Colab to see the bitsandbytes binaries by setting the LD_LIBRARY_PATH\n",
|
| 72 |
+
"# This often resolves the 'latest version' ImportError even when it's already installed\n",
|
| 73 |
+
"import torch\n",
|
| 74 |
+
"\n",
|
| 75 |
+
"def fix_bnb():\n",
|
| 76 |
+
" print(\"Applying bitsandbytes environment fix...\")\n",
|
| 77 |
+
" # Re-install just in case\n",
|
| 78 |
+
" subprocess.run([sys.executable, \"-m\", \"pip\", \"install\", \"-U\", \"bitsandbytes\", \"--quiet\"])\n",
|
| 79 |
+
"\n",
|
| 80 |
+
" # Find the library path\n",
|
| 81 |
+
" import bitsandbytes as bnb\n",
|
| 82 |
+
" print(f\"✓ bnb version: {bnb.__version__}\")\n",
|
| 83 |
+
"\n",
|
| 84 |
+
" # Simple check to see if we can instantiate a 4bit layer\n",
|
| 85 |
+
" try:\n",
|
| 86 |
+
" from transformers import BitsAndBytesConfig\n",
|
| 87 |
+
" test_config = BitsAndBytesConfig(load_in_4bit=True)\n",
|
| 88 |
+
" print(\"✓ Transformers-BNB integration check passed.\")\n",
|
| 89 |
+
" except Exception as e:\n",
|
| 90 |
+
" print(f\"⚠ Integration check failed: {e}\")\n",
|
| 91 |
+
"\n",
|
| 92 |
+
"fix_bnb()"
|
| 93 |
+
]
|
| 94 |
+
},
|
| 95 |
+
{
|
| 96 |
+
"cell_type": "code",
|
| 97 |
+
"execution_count": 3,
|
| 98 |
+
"id": "1eac60b2-6dd1-4869-b641-1eff2408bf49",
|
| 99 |
+
"metadata": {},
|
| 100 |
+
"outputs": [
|
| 101 |
+
{
|
| 102 |
+
"name": "stdout",
|
| 103 |
+
"output_type": "stream",
|
| 104 |
+
"text": [
|
| 105 |
+
"Applying deep patch to bypass bitsandbytes version checks...\n",
|
| 106 |
+
"✓ Transformers internal check successfully bypassed.\n",
|
| 107 |
+
"✓ Current bitsandbytes version: 0.49.2\n",
|
| 108 |
+
"You can now run STEP 6 to load the model.\n"
|
| 109 |
+
]
|
| 110 |
+
}
|
| 111 |
+
],
|
| 112 |
+
"source": [
|
| 113 |
+
"import transformers\n",
|
| 114 |
+
"import bitsandbytes as bnb\n",
|
| 115 |
+
"from transformers.utils import import_utils\n",
|
| 116 |
+
"\n",
|
| 117 |
+
"print(\"Applying deep patch to bypass bitsandbytes version checks...\")\n",
|
| 118 |
+
"\n",
|
| 119 |
+
"try:\n",
|
| 120 |
+
" # 1. Force the internal bitsandbytes availability check to return True\n",
|
| 121 |
+
" import transformers.utils.import_utils as import_utils\n",
|
| 122 |
+
" import_utils.is_bitsandbytes_available = lambda: True\n",
|
| 123 |
+
"\n",
|
| 124 |
+
" # 2. Patch the 4-bit quantizer specifically to skip the version check\n",
|
| 125 |
+
" from transformers.quantizers import quantizer_bnb_4bit\n",
|
| 126 |
+
" quantizer_bnb_4bit.is_bitsandbytes_available = lambda: True\n",
|
| 127 |
+
"\n",
|
| 128 |
+
" # 3. Verify the manual override\n",
|
| 129 |
+
" from transformers import BitsAndBytesConfig\n",
|
| 130 |
+
" test_config = BitsAndBytesConfig(load_in_4bit=True)\n",
|
| 131 |
+
"\n",
|
| 132 |
+
" print(\"✓ Transformers internal check successfully bypassed.\")\n",
|
| 133 |
+
" print(f\"✓ Current bitsandbytes version: {bnb.__version__}\")\n",
|
| 134 |
+
" print(\"You can now run STEP 6 to load the model.\")\n",
|
| 135 |
+
"except Exception as e:\n",
|
| 136 |
+
" print(f\"⚠ Deep patch failed: {e}\")"
|
| 137 |
+
]
|
| 138 |
+
},
|
| 139 |
+
{
|
| 140 |
+
"cell_type": "code",
|
| 141 |
+
"execution_count": 4,
|
| 142 |
+
"id": "0f2fa48e-1321-4836-9006-624c1f127961",
|
| 143 |
+
"metadata": {},
|
| 144 |
+
"outputs": [
|
| 145 |
+
{
|
| 146 |
+
"name": "stdout",
|
| 147 |
+
"output_type": "stream",
|
| 148 |
+
"text": [
|
| 149 |
+
"✓ Environment Restored\n"
|
| 150 |
+
]
|
| 151 |
+
}
|
| 152 |
+
],
|
| 153 |
+
"source": [
|
| 154 |
+
"# Step 2: Global Configuration\n",
|
| 155 |
+
"# Running this to restore variables like MODEL_ID and config settings\n",
|
| 156 |
+
"import json, re, os, warnings\n",
|
| 157 |
+
"import numpy as np\n",
|
| 158 |
+
"import pandas as pd\n",
|
| 159 |
+
"from pathlib import Path\n",
|
| 160 |
+
"from typing import Optional\n",
|
| 161 |
+
"import torch\n",
|
| 162 |
+
"from datasets import Dataset, DatasetDict\n",
|
| 163 |
+
"from transformers import AutoTokenizer, AutoModelForCausalLM, BitsAndBytesConfig, TrainingArguments, EarlyStoppingCallback, set_seed\n",
|
| 164 |
+
"from peft import LoraConfig, TaskType, get_peft_model, PeftModel, prepare_model_for_kbit_training\n",
|
| 165 |
+
"from trl import SFTTrainer\n",
|
| 166 |
+
"from sklearn.metrics import mean_absolute_error\n",
|
| 167 |
+
"from sklearn.model_selection import train_test_split\n",
|
| 168 |
+
"\n",
|
| 169 |
+
"warnings.filterwarnings('ignore')\n",
|
| 170 |
+
"set_seed(42)\n",
|
| 171 |
+
"\n",
|
| 172 |
+
"MODEL_ID = 'Qwen/Qwen2.5-7B-Instruct'\n",
|
| 173 |
+
"ADAPTER_DIR = './pels_qlora_adapter'\n",
|
| 174 |
+
"JSONL_PATH = './pels_dataset.jsonl'\n",
|
| 175 |
+
"EXCEL_PATH = 'PELS_Final_Justification.xlsx'\n",
|
| 176 |
+
"\n",
|
| 177 |
+
"SCORE_COLS_RAW = ['C1\\nFoundations', 'C2\\nDesign', 'C3\\nOutput Spec', 'C4\\nDomain', 'C5\\nEthics', 'C6\\nMeta-\\ncognition', 'Final\\nScore']\n",
|
| 178 |
+
"SCORE_KEYS = ['C1', 'C2', 'C3', 'C4', 'C5', 'C6', 'final_score']\n",
|
| 179 |
+
"RUBRIC_WEIGHTS = {'C1': 0.15, 'C2': 0.20, 'C3': 0.20, 'C4': 0.20, 'C5': 0.15, 'C6': 0.10}\n",
|
| 180 |
+
"\n",
|
| 181 |
+
"MAX_SEQ_LEN = 2048\n",
|
| 182 |
+
"BATCH_SIZE = 2\n",
|
| 183 |
+
"GRAD_ACCUM = 4\n",
|
| 184 |
+
"LR = 2e-4\n",
|
| 185 |
+
"EPOCHS = 3\n",
|
| 186 |
+
"LORA_R = 16\n",
|
| 187 |
+
"LORA_ALPHA = 32\n",
|
| 188 |
+
"LORA_DROPOUT = 0.05\n",
|
| 189 |
+
"\n",
|
| 190 |
+
"print('✓ Environment Restored')"
|
| 191 |
+
]
|
| 192 |
+
},
|
| 193 |
+
{
|
| 194 |
+
"cell_type": "code",
|
| 195 |
+
"execution_count": null,
|
| 196 |
+
"id": "9ad5c533-3c40-4dfd-8cf8-39c65a20f679",
|
| 197 |
+
"metadata": {},
|
| 198 |
+
"outputs": [],
|
| 199 |
+
"source": []
|
| 200 |
+
}
|
| 201 |
+
],
|
| 202 |
+
"metadata": {
|
| 203 |
+
"kernelspec": {
|
| 204 |
+
"display_name": "SC (GPU-Fixed)",
|
| 205 |
+
"language": "python",
|
| 206 |
+
"name": "softcomp"
|
| 207 |
+
},
|
| 208 |
+
"language_info": {
|
| 209 |
+
"codemirror_mode": {
|
| 210 |
+
"name": "ipython",
|
| 211 |
+
"version": 3
|
| 212 |
+
},
|
| 213 |
+
"file_extension": ".py",
|
| 214 |
+
"mimetype": "text/x-python",
|
| 215 |
+
"name": "python",
|
| 216 |
+
"nbconvert_exporter": "python",
|
| 217 |
+
"pygments_lexer": "ipython3",
|
| 218 |
+
"version": "3.10.20"
|
| 219 |
+
}
|
| 220 |
+
},
|
| 221 |
+
"nbformat": 4,
|
| 222 |
+
"nbformat_minor": 5
|
| 223 |
+
}
|
app.py
ADDED
|
@@ -0,0 +1,501 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# ── Cell: PELS Gradio App (Scenario-Based) ────────────────────────────────────
|
| 2 |
+
|
| 3 |
+
!pip install gradio -q
|
| 4 |
+
|
| 5 |
+
import gradio as gr
|
| 6 |
+
import torch
|
| 7 |
+
import re
|
| 8 |
+
import random
|
| 9 |
+
|
| 10 |
+
model.eval()
|
| 11 |
+
torch.cuda.empty_cache()
|
| 12 |
+
|
| 13 |
+
RUBRIC = """C1 Foundations (15%): Task clarity, role setup, AI awareness.
|
| 14 |
+
C2 Design (20%): Prompt structure, patterns (few-shot, CoT, role+task+constraint).
|
| 15 |
+
C3 Output Spec (20%): Format, length, tone, structure constraints.
|
| 16 |
+
C4 Domain Application (20%): Domain vocabulary, contextual accuracy.
|
| 17 |
+
C5 Ethics (15%): No harmful/biased framing. Score=1 triggers automatic Final Score override to 1.0.
|
| 18 |
+
C6 Metacognition (10%): Self-awareness, iteration design, fallback handling."""
|
| 19 |
+
|
| 20 |
+
MAX_LENGTH = 512
|
| 21 |
+
MAX_NEW_TOKENS = 150
|
| 22 |
+
|
| 23 |
+
# ── Scenario bank ─────────────────────────────────────────────────────────────
|
| 24 |
+
# Each scenario has:
|
| 25 |
+
# "scenario" — the situation shown to the user
|
| 26 |
+
# "task" — what they are asked to do (no hints on HOW to prompt)
|
| 27 |
+
# The user must figure out how to write the prompt themselves.
|
| 28 |
+
|
| 29 |
+
SCENARIOS = {
|
| 30 |
+
"CAREER": [
|
| 31 |
+
{
|
| 32 |
+
"scenario": "Rohan is a mechanical engineer with 5 years of experience who wants to switch into data science. He has done one online Python course but has no real projects yet. He has an interview at a data analytics firm next month.",
|
| 33 |
+
"task": "Write an AI prompt that helps Rohan prepare for this career transition.",
|
| 34 |
+
},
|
| 35 |
+
{
|
| 36 |
+
"scenario": "Priya has been a school teacher for 8 years and wants to move into corporate L&D (Learning & Development). She has no corporate experience but has designed curriculum and trained 200+ students.",
|
| 37 |
+
"task": "Write an AI prompt that helps Priya position herself for an L&D role.",
|
| 38 |
+
},
|
| 39 |
+
{
|
| 40 |
+
"scenario": "Amir graduated 6 months ago with a BCA degree and has been applying for software developer roles but getting no callbacks. His resume lists his college projects but no internships.",
|
| 41 |
+
"task": "Write an AI prompt that helps Amir fix the problem and get more callbacks.",
|
| 42 |
+
},
|
| 43 |
+
{
|
| 44 |
+
"scenario": "Sneha is a marketing manager at a mid-size company. She wants to ask for a promotion to Director but has never negotiated salary or title before and doesn't know how to make the case.",
|
| 45 |
+
"task": "Write an AI prompt that helps Sneha prepare for this conversation with her manager.",
|
| 46 |
+
},
|
| 47 |
+
],
|
| 48 |
+
"EDUCATION": [
|
| 49 |
+
{
|
| 50 |
+
"scenario": "A Grade 9 teacher needs to explain the concept of compound interest to students who understand basic multiplication and percentages but have never studied finance or banking.",
|
| 51 |
+
"task": "Write an AI prompt that produces a teaching resource for this class.",
|
| 52 |
+
},
|
| 53 |
+
{
|
| 54 |
+
"scenario": "A university professor wants to check whether her 2nd-year engineering students have understood Newton's Laws of Motion — specifically common misconceptions students have about inertia.",
|
| 55 |
+
"task": "Write an AI prompt that generates an assessment to test this understanding.",
|
| 56 |
+
},
|
| 57 |
+
{
|
| 58 |
+
"scenario": "A homeschooling parent needs to teach their 10-year-old child about climate change in a way that is factually accurate but not scary or overwhelming, using everyday examples.",
|
| 59 |
+
"task": "Write an AI prompt that creates an age-appropriate lesson on this topic.",
|
| 60 |
+
},
|
| 61 |
+
{
|
| 62 |
+
"scenario": "A coding bootcamp instructor wants to introduce recursion to students who are comfortable with loops (for/while) but have never seen a function call itself.",
|
| 63 |
+
"task": "Write an AI prompt that creates a beginner-friendly explanation with an exercise.",
|
| 64 |
+
},
|
| 65 |
+
],
|
| 66 |
+
"TECHNOLOGY": [
|
| 67 |
+
{
|
| 68 |
+
"scenario": "A junior developer at a startup wrote a Python script that reads a CSV file and calculates monthly sales totals — but it crashes whenever a cell is empty or contains text instead of a number.",
|
| 69 |
+
"task": "Write an AI prompt that helps fix and improve this script.",
|
| 70 |
+
},
|
| 71 |
+
{
|
| 72 |
+
"scenario": "A non-technical product manager needs to explain to her team why their app is slow. The engineering team says it is a 'database N+1 query problem' but she doesn't understand what that means.",
|
| 73 |
+
"task": "Write an AI prompt that produces an explanation she can actually understand and relay to stakeholders.",
|
| 74 |
+
},
|
| 75 |
+
{
|
| 76 |
+
"scenario": "A small business owner wants to build a simple website contact form that stores submissions in a Google Sheet — they have no coding experience and a budget of zero.",
|
| 77 |
+
"task": "Write an AI prompt that gives them a practical, step-by-step solution.",
|
| 78 |
+
},
|
| 79 |
+
{
|
| 80 |
+
"scenario": "A data analyst has a pandas DataFrame with 500,000 rows. Her current code takes 4 minutes to run a groupby aggregation. Her manager wants results in under 30 seconds.",
|
| 81 |
+
"task": "Write an AI prompt that helps her optimise the code.",
|
| 82 |
+
},
|
| 83 |
+
],
|
| 84 |
+
"HEALTHCARE": [
|
| 85 |
+
{
|
| 86 |
+
"scenario": "A 45-year-old patient was just diagnosed with pre-diabetes. Their doctor told them to 'watch their diet and exercise more' but gave no specific guidance. The patient is confused about what to actually do.",
|
| 87 |
+
"task": "Write an AI prompt that produces practical, safe guidance for this patient.",
|
| 88 |
+
},
|
| 89 |
+
{
|
| 90 |
+
"scenario": "A nurse manager at a clinic needs to train new staff on the correct procedure for hand hygiene according to WHO guidelines — in a way that is quick to read and easy to remember during a busy shift.",
|
| 91 |
+
"task": "Write an AI prompt that creates this training material.",
|
| 92 |
+
},
|
| 93 |
+
{
|
| 94 |
+
"scenario": "A medical student is struggling to remember the differences between Type 1 and Type 2 Diabetes — the symptoms, causes, treatment approaches, and which patient populations are typically affected.",
|
| 95 |
+
"task": "Write an AI prompt that creates a study aid for this topic.",
|
| 96 |
+
},
|
| 97 |
+
{
|
| 98 |
+
"scenario": "A hospital wants to send a clear, non-alarming message to patients reminding them to get their annual flu vaccination — the message will go out via SMS so it must be very short.",
|
| 99 |
+
"task": "Write an AI prompt that generates this patient communication.",
|
| 100 |
+
},
|
| 101 |
+
],
|
| 102 |
+
"LEGAL": [
|
| 103 |
+
{
|
| 104 |
+
"scenario": "A freelance graphic designer in Pune completed a logo project for a client who is now refusing to pay the ₹25,000 invoice, claiming the work was 'not what was agreed'. There was a WhatsApp conversation but no formal contract.",
|
| 105 |
+
"task": "Write an AI prompt that helps the designer understand their options and next steps.",
|
| 106 |
+
},
|
| 107 |
+
{
|
| 108 |
+
"scenario": "A first-time landlord in Bangalore wants to rent out their apartment. They have heard that verbal agreements can cause problems and want to create a proper rental agreement that protects them legally.",
|
| 109 |
+
"task": "Write an AI prompt that helps them draft or understand what should be in this agreement.",
|
| 110 |
+
},
|
| 111 |
+
{
|
| 112 |
+
"scenario": "An employee received a termination letter from their company citing 'performance issues' but believes they were fired because they filed a complaint against their manager last month.",
|
| 113 |
+
"task": "Write an AI prompt that helps this person understand whether they have a case and what to do.",
|
| 114 |
+
},
|
| 115 |
+
{
|
| 116 |
+
"scenario": "A startup founder is about to sign a 3-year office lease. They have never signed a commercial lease before and are worried about clauses that could trap them if the startup fails in year one.",
|
| 117 |
+
"task": "Write an AI prompt that helps them know what to watch out for in this agreement.",
|
| 118 |
+
},
|
| 119 |
+
],
|
| 120 |
+
"MARKETING": [
|
| 121 |
+
{
|
| 122 |
+
"scenario": "A local bakery in Hyderabad has great reviews but almost no online presence. They want to attract customers aged 20–35 who discover food businesses on Instagram. Their budget is zero — only organic content.",
|
| 123 |
+
"task": "Write an AI prompt that helps them create a content strategy or specific post.",
|
| 124 |
+
},
|
| 125 |
+
{
|
| 126 |
+
"scenario": "An edtech startup is launching a new course on AI for non-technical professionals. They need to send a launch email to their existing subscriber list of 5,000 people — most of whom haven't opened emails in 3 months.",
|
| 127 |
+
"task": "Write an AI prompt that produces this re-engagement launch email.",
|
| 128 |
+
},
|
| 129 |
+
{
|
| 130 |
+
"scenario": "A fitness trainer wants to run Google Ads for their personal training services in Mumbai. They have a ₹10,000/month budget and have never run paid ads before. Their USP is online coaching with personalised meal plans.",
|
| 131 |
+
"task": "Write an AI prompt that helps them set up or plan this campaign effectively.",
|
| 132 |
+
},
|
| 133 |
+
{
|
| 134 |
+
"scenario": "A sustainable clothing brand is launching a new line made from recycled ocean plastic. Their target customer cares about the environment but is price-sensitive (products are 30% more expensive than fast fashion).",
|
| 135 |
+
"task": "Write an AI prompt that creates compelling product description copy for their website.",
|
| 136 |
+
},
|
| 137 |
+
],
|
| 138 |
+
"FINANCE": [
|
| 139 |
+
{
|
| 140 |
+
"scenario": "A 28-year-old software developer earns ₹1.2 lakh per month but saves almost nothing. They have ₹3 lakh in credit card debt at 36% annual interest and no investments. They want to start getting their finances in order.",
|
| 141 |
+
"task": "Write an AI prompt that produces a practical financial plan for this person.",
|
| 142 |
+
},
|
| 143 |
+
{
|
| 144 |
+
"scenario": "A small business owner wants to understand their company's cash flow statement. Their accountant gave them a document but they don't understand why the business is profitable on paper but always short on cash.",
|
| 145 |
+
"task": "Write an AI prompt that explains this concept in a way they can immediately apply to their situation.",
|
| 146 |
+
},
|
| 147 |
+
{
|
| 148 |
+
"scenario": "A couple wants to save for their child's higher education. The child is currently 3 years old and they estimate they'll need ₹30 lakhs in 15 years. They can invest ₹10,000 per month.",
|
| 149 |
+
"task": "Write an AI prompt that helps them understand their investment options and whether their goal is achievable.",
|
| 150 |
+
},
|
| 151 |
+
{
|
| 152 |
+
"scenario": "A salaried employee received their first Form 16 and has to file their ITR for the first time. They are confused about which ITR form to use, what deductions they can claim under 80C, and how to avoid mistakes.",
|
| 153 |
+
"task": "Write an AI prompt that guides them through this process.",
|
| 154 |
+
},
|
| 155 |
+
],
|
| 156 |
+
"CREATIVE": [
|
| 157 |
+
{
|
| 158 |
+
"scenario": "A screenwriter wants to write a 5-minute short film about loneliness in a big city. The protagonist is a 30-year-old delivery driver who interacts with dozens of people every day but has no real relationships.",
|
| 159 |
+
"task": "Write an AI prompt that helps develop this concept into a concrete scene or script outline.",
|
| 160 |
+
},
|
| 161 |
+
{
|
| 162 |
+
"scenario": "A startup founder needs to write the 'About Us' page for their company website. The company builds AI tools for teachers. The tone should feel human and mission-driven, not corporate or salesy.",
|
| 163 |
+
"task": "Write an AI prompt that produces this About Us page.",
|
| 164 |
+
},
|
| 165 |
+
{
|
| 166 |
+
"scenario": "A children's book author wants to write a short story for 6–8 year olds that teaches them about the importance of asking for help — without being preachy. The story should have an animal character.",
|
| 167 |
+
"task": "Write an AI prompt that generates this story or a strong outline for it.",
|
| 168 |
+
},
|
| 169 |
+
{
|
| 170 |
+
"scenario": "A musician wants to write lyrics for an indie-folk song about their grandmother who passed away last year. The mood should be bittersweet — celebrating her life rather than mourning — with imagery from her kitchen and garden.",
|
| 171 |
+
"task": "Write an AI prompt that helps generate these lyrics or a draft verse.",
|
| 172 |
+
},
|
| 173 |
+
],
|
| 174 |
+
}
|
| 175 |
+
|
| 176 |
+
# ── Core functions ────────────────────────────────────────────────────────────
|
| 177 |
+
def get_scenario(domain):
|
| 178 |
+
if not domain:
|
| 179 |
+
return "", ""
|
| 180 |
+
scenarios = SCENARIOS.get(domain, [])
|
| 181 |
+
if not scenarios:
|
| 182 |
+
return "", ""
|
| 183 |
+
picked = random.choice(scenarios)
|
| 184 |
+
scenario_html = f"""
|
| 185 |
+
<div style="background:#0f172a;border-radius:10px;padding:20px 22px;
|
| 186 |
+
border-left:4px solid #6366f1;font-family:sans-serif">
|
| 187 |
+
<div style="color:#a5b4fc;font-size:11px;font-weight:700;
|
| 188 |
+
letter-spacing:.1em;margin-bottom:10px">
|
| 189 |
+
📋 YOUR SCENARIO
|
| 190 |
+
</div>
|
| 191 |
+
<div style="color:#e2e8f0;font-size:15px;line-height:1.75;margin-bottom:16px">
|
| 192 |
+
{picked['scenario']}
|
| 193 |
+
</div>
|
| 194 |
+
<div style="background:#1e293b;border-radius:8px;padding:12px 16px">
|
| 195 |
+
<span style="color:#fbbf24;font-weight:700;font-size:13px">🎯 Your Task: </span>
|
| 196 |
+
<span style="color:#fde68a;font-size:14px">{picked['task']}</span>
|
| 197 |
+
</div>
|
| 198 |
+
<div style="margin-top:14px;color:#475569;font-size:12px;font-style:italic">
|
| 199 |
+
💡 Tip: A strong prompt assigns a role to the AI, specifies the audience,
|
| 200 |
+
defines the output format, and sets constraints. You figure out how — that's the assessment.
|
| 201 |
+
</div>
|
| 202 |
+
</div>"""
|
| 203 |
+
return scenario_html, picked["scenario"] + " | Task: " + picked["task"]
|
| 204 |
+
|
| 205 |
+
def get_tier(score):
|
| 206 |
+
"""score is always 0–10"""
|
| 207 |
+
if score >= 7.1: return "🟢 Strong", "#22c55e"
|
| 208 |
+
elif score >= 4.1: return "🟡 Developing", "#f59e0b"
|
| 209 |
+
else: return "🔴 Weak", "#ef4444"
|
| 210 |
+
|
| 211 |
+
def extract_scores(text):
|
| 212 |
+
patterns = {
|
| 213 |
+
'c1': r'C1_Foundations:\s*([0-9.]+)',
|
| 214 |
+
'c2': r'C2_Design:\s*([0-9.]+)',
|
| 215 |
+
'c3': r'C3_OutputSpec:\s*([0-9.]+)',
|
| 216 |
+
'c4': r'C4_Domain:\s*([0-9.]+)',
|
| 217 |
+
'c5': r'C5_Ethics:\s*([0-9.]+)',
|
| 218 |
+
'c6': r'C6_Metacognition:\s*([0-9.]+)',
|
| 219 |
+
'final': r'Final_Score:\s*([0-9.]+)',
|
| 220 |
+
}
|
| 221 |
+
return {k: float(m.group(1)) if (m := re.search(p, text)) else None
|
| 222 |
+
for k, p in patterns.items()}
|
| 223 |
+
|
| 224 |
+
def grade_prompt_fast(prompt_text, domain):
|
| 225 |
+
input_text = (
|
| 226 |
+
f"Instruct: ### PELS Grading Task\n"
|
| 227 |
+
f"Domain: {domain}\n"
|
| 228 |
+
f"Rubric:\n{RUBRIC}\n\n"
|
| 229 |
+
f"Candidate Prompt:\n{prompt_text}\n\n"
|
| 230 |
+
f"### Evaluation\n"
|
| 231 |
+
f"Score each category 1–10. Ethics score of 1 overrides all others.\n"
|
| 232 |
+
f"Output:"
|
| 233 |
+
)
|
| 234 |
+
inputs = tokenizer(
|
| 235 |
+
input_text, return_tensors='pt',
|
| 236 |
+
truncation=True, max_length=MAX_LENGTH
|
| 237 |
+
).to(model.device)
|
| 238 |
+
|
| 239 |
+
with torch.no_grad():
|
| 240 |
+
outputs = model.generate(
|
| 241 |
+
**inputs,
|
| 242 |
+
max_new_tokens=MAX_NEW_TOKENS,
|
| 243 |
+
do_sample=False,
|
| 244 |
+
pad_token_id=tokenizer.eos_token_id,
|
| 245 |
+
)
|
| 246 |
+
decoded = tokenizer.decode(outputs[0], skip_special_tokens=True)
|
| 247 |
+
out_part = decoded[len(input_text):] if input_text in decoded else decoded
|
| 248 |
+
return out_part, extract_scores(out_part)
|
| 249 |
+
|
| 250 |
+
def build_score_html(scores, veto, raw_output):
|
| 251 |
+
final = scores.get("final") or 0.0
|
| 252 |
+
final_display = final if final > 1.0 else final * 10
|
| 253 |
+
bar_pct = min(int(final_display * 10), 100)
|
| 254 |
+
t_label, color = get_tier(final_display)
|
| 255 |
+
|
| 256 |
+
cat_info = [
|
| 257 |
+
("c1", "C1 · Foundations", "15%"),
|
| 258 |
+
("c2", "C2 · Design", "20%"),
|
| 259 |
+
("c3", "C3 · Output Spec", "20%"),
|
| 260 |
+
("c4", "C4 · Domain Application", "20%"),
|
| 261 |
+
("c5", "C5 · Ethics", "15%"),
|
| 262 |
+
("c6", "C6 · Metacognition", "10%"),
|
| 263 |
+
]
|
| 264 |
+
|
| 265 |
+
rows = ""
|
| 266 |
+
for key, label, weight in cat_info:
|
| 267 |
+
val = scores.get(key)
|
| 268 |
+
if val is None:
|
| 269 |
+
rows += f"""<tr>
|
| 270 |
+
<td style="padding:8px 12px;color:#94a3b8;font-family:sans-serif;
|
| 271 |
+
white-space:nowrap">{label} <span style="color:#334155">({weight})</span></td>
|
| 272 |
+
<td colspan="2" style="color:#475569;padding:8px 12px">—</td>
|
| 273 |
+
</tr>"""
|
| 274 |
+
continue
|
| 275 |
+
val_d = val if val > 1.0 else val * 10
|
| 276 |
+
_, c = get_tier(val_d)
|
| 277 |
+
pct = min(int(val_d * 10), 100)
|
| 278 |
+
rows += f"""<tr>
|
| 279 |
+
<td style="padding:8px 12px;font-weight:600;color:#cbd5e1;
|
| 280 |
+
white-space:nowrap;font-family:sans-serif">
|
| 281 |
+
{label} <span style="color:#475569;font-weight:400">({weight})</span>
|
| 282 |
+
</td>
|
| 283 |
+
<td style="padding:8px 12px;width:100%">
|
| 284 |
+
<div style="background:#1e293b;border-radius:4px;height:13px;overflow:hidden">
|
| 285 |
+
<div style="background:{c};width:{pct}%;height:100%;border-radius:4px"></div>
|
| 286 |
+
</div>
|
| 287 |
+
</td>
|
| 288 |
+
<td style="padding:8px 12px;font-weight:700;color:{c};
|
| 289 |
+
white-space:nowrap;font-family:sans-serif">{val_d:.1f}/10</td>
|
| 290 |
+
</tr>"""
|
| 291 |
+
|
| 292 |
+
veto_banner = ""
|
| 293 |
+
if veto:
|
| 294 |
+
veto_banner = """<div style="background:#7f1d1d;border:1px solid #dc2626;
|
| 295 |
+
border-radius:8px;padding:10px 16px;margin-bottom:16px;
|
| 296 |
+
color:#fca5a5;font-weight:600;font-family:sans-serif">
|
| 297 |
+
⛔ Ethics Veto Triggered — Final Score overridden to 1.0
|
| 298 |
+
</div>"""
|
| 299 |
+
|
| 300 |
+
just_match = re.search(r'Justification:\s*(.+)', raw_output, re.DOTALL)
|
| 301 |
+
justification = just_match.group(1).strip()[:500] if just_match else ""
|
| 302 |
+
just_html = ""
|
| 303 |
+
if justification:
|
| 304 |
+
just_html = f"""
|
| 305 |
+
<div style="margin-top:20px;padding:14px 16px;background:#0f172a;
|
| 306 |
+
border-left:3px solid #6366f1;border-radius:0 8px 8px 0">
|
| 307 |
+
<div style="color:#a5b4fc;font-size:11px;font-weight:700;
|
| 308 |
+
letter-spacing:.08em;margin-bottom:8px;font-family:sans-serif">
|
| 309 |
+
JUSTIFICATION
|
| 310 |
+
</div>
|
| 311 |
+
<div style="color:#cbd5e1;line-height:1.7;font-size:14px;font-family:sans-serif">
|
| 312 |
+
{justification}
|
| 313 |
+
</div>
|
| 314 |
+
</div>"""
|
| 315 |
+
|
| 316 |
+
return f"""
|
| 317 |
+
<div style="font-family:sans-serif;background:#0f172a;border-radius:12px;
|
| 318 |
+
padding:24px;color:#e2e8f0">
|
| 319 |
+
{veto_banner}
|
| 320 |
+
<div style="display:flex;align-items:center;gap:24px;margin-bottom:24px">
|
| 321 |
+
<div style="text-align:center;min-width:90px">
|
| 322 |
+
<div style="font-size:54px;font-weight:800;color:{color};line-height:1">
|
| 323 |
+
{final_display:.1f}
|
| 324 |
+
</div>
|
| 325 |
+
<div style="font-size:12px;color:#64748b;margin-top:4px">out of 10</div>
|
| 326 |
+
</div>
|
| 327 |
+
<div style="flex:1">
|
| 328 |
+
<div style="background:#1e293b;border-radius:8px;height:20px;overflow:hidden">
|
| 329 |
+
<div style="background:{color};width:{bar_pct}%;height:100%;border-radius:8px">
|
| 330 |
+
</div>
|
| 331 |
+
</div>
|
| 332 |
+
<div style="margin-top:10px;font-size:17px;font-weight:700;color:{color}">
|
| 333 |
+
{t_label}
|
| 334 |
+
</div>
|
| 335 |
+
<div style="font-size:12px;color:#475569;margin-top:3px">PELS Final Score</div>
|
| 336 |
+
</div>
|
| 337 |
+
</div>
|
| 338 |
+
<table style="width:100%;border-collapse:collapse;font-size:14px">{rows}</table>
|
| 339 |
+
{just_html}
|
| 340 |
+
</div>"""
|
| 341 |
+
|
| 342 |
+
# ── Handlers ──────────────────────────────────────────────────────────────────
|
| 343 |
+
def on_domain_change(domain):
|
| 344 |
+
html, context = get_scenario(domain)
|
| 345 |
+
return html, context, "", PLACEHOLDER
|
| 346 |
+
|
| 347 |
+
def on_new_scenario(domain):
|
| 348 |
+
html, context = get_scenario(domain)
|
| 349 |
+
return html, context, "", PLACEHOLDER
|
| 350 |
+
|
| 351 |
+
def evaluate(domain, user_prompt, scenario_context):
|
| 352 |
+
if not domain:
|
| 353 |
+
return "<p style='color:#f87171;font-family:sans-serif'>⚠️ Please select a domain first.</p>"
|
| 354 |
+
if not user_prompt or len(user_prompt.strip()) < 10:
|
| 355 |
+
return "<p style='color:#f87171;font-family:sans-serif'>⚠️ Please write your prompt (at least 10 characters).</p>"
|
| 356 |
+
try:
|
| 357 |
+
raw, scores = grade_prompt_fast(user_prompt.strip(), domain)
|
| 358 |
+
except Exception as e:
|
| 359 |
+
return f"<p style='color:#f87171;font-family:sans-serif'>❌ Error: {e}</p>"
|
| 360 |
+
|
| 361 |
+
if not scores or all(v is None for v in scores.values()):
|
| 362 |
+
return f"""<div style='padding:16px;background:#1e293b;border-radius:8px;
|
| 363 |
+
color:#cbd5e1;font-family:sans-serif'>
|
| 364 |
+
<b>Raw model output:</b><br>
|
| 365 |
+
<pre style='white-space:pre-wrap;font-size:12px;color:#94a3b8'>{raw[:600]}</pre>
|
| 366 |
+
<small style='color:#64748b'>Scores not parsed. Try a more detailed prompt.</small>
|
| 367 |
+
</div>"""
|
| 368 |
+
|
| 369 |
+
veto = (scores.get("c5") or 10) <= 1.0
|
| 370 |
+
if veto:
|
| 371 |
+
scores["final"] = 1.0
|
| 372 |
+
|
| 373 |
+
return build_score_html(scores, veto, raw)
|
| 374 |
+
|
| 375 |
+
def clear_all():
|
| 376 |
+
return None, "", "", PLACEHOLDER
|
| 377 |
+
|
| 378 |
+
PLACEHOLDER = """
|
| 379 |
+
<div style='background:#0f172a;border-radius:12px;padding:40px;
|
| 380 |
+
text-align:center;color:#334155;font-family:sans-serif;font-size:15px'>
|
| 381 |
+
Your PELS score report will appear here after evaluation.
|
| 382 |
+
</div>"""
|
| 383 |
+
|
| 384 |
+
# ── UI ────────────────────────────────────────────────────────────────────────
|
| 385 |
+
with gr.Blocks(title="PELS Prompt Assessment", theme=gr.themes.Base()) as demo:
|
| 386 |
+
|
| 387 |
+
scenario_context = gr.State("") # hidden state stores scenario text
|
| 388 |
+
|
| 389 |
+
gr.HTML("""
|
| 390 |
+
<div style="padding:24px 0 12px">
|
| 391 |
+
<h1 style="font-family:sans-serif;font-size:2rem;font-weight:800;margin:0;
|
| 392 |
+
background:linear-gradient(135deg,#6366f1,#a855f7,#ec4899);
|
| 393 |
+
-webkit-background-clip:text;-webkit-text-fill-color:transparent">
|
| 394 |
+
PELS Prompt Assessment
|
| 395 |
+
</h1>
|
| 396 |
+
<p style="font-family:sans-serif;color:#64748b;margin:6px 0 0;font-size:14px">
|
| 397 |
+
You will receive a real-world scenario. Write an AI prompt to address it.
|
| 398 |
+
Your prompt will be evaluated across 6 rubric categories.
|
| 399 |
+
</p>
|
| 400 |
+
</div>""")
|
| 401 |
+
|
| 402 |
+
with gr.Row():
|
| 403 |
+
|
| 404 |
+
# ── Left ──────────────────────────────────────────────────────────────
|
| 405 |
+
with gr.Column(scale=1):
|
| 406 |
+
|
| 407 |
+
gr.HTML('<div style="font-family:sans-serif;font-weight:700;'
|
| 408 |
+
'color:#a5b4fc;font-size:12px;letter-spacing:.08em;'
|
| 409 |
+
'margin-bottom:6px">STEP 1 · CHOOSE DOMAIN</div>')
|
| 410 |
+
|
| 411 |
+
domain_dd = gr.Dropdown(
|
| 412 |
+
choices=list(SCENARIOS.keys()),
|
| 413 |
+
label="Domain", value=None, interactive=True,
|
| 414 |
+
)
|
| 415 |
+
|
| 416 |
+
gr.HTML('<div style="font-family:sans-serif;font-weight:700;'
|
| 417 |
+
'color:#a5b4fc;font-size:12px;letter-spacing:.08em;'
|
| 418 |
+
'margin:18px 0 6px">STEP 2 · YOUR SCENARIO</div>')
|
| 419 |
+
|
| 420 |
+
scenario_display = gr.HTML(
|
| 421 |
+
value="""<div style='background:#0f172a;border-radius:10px;
|
| 422 |
+
padding:20px;text-align:center;color:#334155;
|
| 423 |
+
font-family:sans-serif'>
|
| 424 |
+
Select a domain above to receive your scenario.
|
| 425 |
+
</div>"""
|
| 426 |
+
)
|
| 427 |
+
|
| 428 |
+
new_scenario_btn = gr.Button(
|
| 429 |
+
"🔀 Get Different Scenario", variant="secondary", size="sm"
|
| 430 |
+
)
|
| 431 |
+
|
| 432 |
+
gr.HTML('<div style="font-family:sans-serif;font-weight:700;'
|
| 433 |
+
'color:#a5b4fc;font-size:12px;letter-spacing:.08em;'
|
| 434 |
+
'margin:18px 0 6px">STEP 3 · WRITE YOUR PROMPT</div>')
|
| 435 |
+
|
| 436 |
+
prompt_box = gr.Textbox(
|
| 437 |
+
label="Your AI Prompt",
|
| 438 |
+
placeholder="Based on the scenario above, write your AI prompt here…",
|
| 439 |
+
lines=9,
|
| 440 |
+
)
|
| 441 |
+
|
| 442 |
+
with gr.Row():
|
| 443 |
+
clear_btn = gr.Button("🗑 Clear", variant="secondary", size="sm")
|
| 444 |
+
eval_btn = gr.Button("⚡ Evaluate", variant="primary", size="lg")
|
| 445 |
+
|
| 446 |
+
gr.HTML("""
|
| 447 |
+
<div style="margin-top:14px;padding:12px 14px;background:#1e293b;
|
| 448 |
+
border-radius:8px;font-size:12px;color:#64748b;
|
| 449 |
+
font-family:sans-serif;line-height:1.8">
|
| 450 |
+
<b style="color:#94a3b8">Rubric weights:</b><br>
|
| 451 |
+
C1 Foundations 15% · C2 Design 20% · C3 Output Spec 20%<br>
|
| 452 |
+
C4 Domain 20% · C5 Ethics 15% · C6 Metacognition 10%<br><br>
|
| 453 |
+
<b style="color:#94a3b8">Tiers:</b>
|
| 454 |
+
🟢 Strong ≥ 7.1
|
| 455 |
+
· 🟡 Developing 4.1–7.0
|
| 456 |
+
· 🔴 Weak ≤ 4.0
|
| 457 |
+
</div>""")
|
| 458 |
+
|
| 459 |
+
# ── Right ────��────────────────────────────────────────────────────────
|
| 460 |
+
with gr.Column(scale=1):
|
| 461 |
+
|
| 462 |
+
gr.HTML('<div style="font-family:sans-serif;font-weight:700;'
|
| 463 |
+
'color:#a5b4fc;font-size:12px;letter-spacing:.08em;'
|
| 464 |
+
'margin-bottom:6px">STEP 4 · PELS SCORE REPORT</div>')
|
| 465 |
+
|
| 466 |
+
result_html = gr.HTML(value=PLACEHOLDER)
|
| 467 |
+
|
| 468 |
+
with gr.Accordion("📖 Rubric Reference", open=False):
|
| 469 |
+
gr.Markdown("""
|
| 470 |
+
| Category | Weight | What it checks |
|
| 471 |
+
|---|---|---|
|
| 472 |
+
| C1 Foundations | 15% | Task clarity, role setup, AI awareness |
|
| 473 |
+
| C2 Design | 20% | Prompt structure — few-shot, CoT, role+task+constraint patterns |
|
| 474 |
+
| C3 Output Spec | 20% | Format, length, tone, structure constraints |
|
| 475 |
+
| C4 Domain | 20% | Domain vocabulary and contextual accuracy |
|
| 476 |
+
| C5 Ethics | 15% | No harmful or biased framing — score of 1 overrides the final score |
|
| 477 |
+
| C6 Metacognition | 10% | Self-awareness, iteration design, fallback handling |
|
| 478 |
+
""")
|
| 479 |
+
|
| 480 |
+
# ── Events ────────────────────────────────────────────────────────────────
|
| 481 |
+
domain_dd.change(
|
| 482 |
+
fn=on_domain_change,
|
| 483 |
+
inputs=domain_dd,
|
| 484 |
+
outputs=[scenario_display, scenario_context, prompt_box, result_html]
|
| 485 |
+
)
|
| 486 |
+
new_scenario_btn.click(
|
| 487 |
+
fn=on_new_scenario,
|
| 488 |
+
inputs=domain_dd,
|
| 489 |
+
outputs=[scenario_display, scenario_context, prompt_box, result_html]
|
| 490 |
+
)
|
| 491 |
+
eval_btn.click(
|
| 492 |
+
fn=evaluate,
|
| 493 |
+
inputs=[domain_dd, prompt_box, scenario_context],
|
| 494 |
+
outputs=result_html
|
| 495 |
+
)
|
| 496 |
+
clear_btn.click(
|
| 497 |
+
fn=clear_all,
|
| 498 |
+
outputs=[domain_dd, scenario_display, prompt_box, result_html]
|
| 499 |
+
)
|
| 500 |
+
|
| 501 |
+
demo.launch(share=True)
|
pels_correlation.png
ADDED
|
pels_phi2_qlora/README.md
ADDED
|
@@ -0,0 +1,202 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
base_model: microsoft/phi-2
|
| 3 |
+
library_name: peft
|
| 4 |
+
---
|
| 5 |
+
|
| 6 |
+
# Model Card for Model ID
|
| 7 |
+
|
| 8 |
+
<!-- Provide a quick summary of what the model is/does. -->
|
| 9 |
+
|
| 10 |
+
|
| 11 |
+
|
| 12 |
+
## Model Details
|
| 13 |
+
|
| 14 |
+
### Model Description
|
| 15 |
+
|
| 16 |
+
<!-- Provide a longer summary of what this model is. -->
|
| 17 |
+
|
| 18 |
+
|
| 19 |
+
|
| 20 |
+
- **Developed by:** [More Information Needed]
|
| 21 |
+
- **Funded by [optional]:** [More Information Needed]
|
| 22 |
+
- **Shared by [optional]:** [More Information Needed]
|
| 23 |
+
- **Model type:** [More Information Needed]
|
| 24 |
+
- **Language(s) (NLP):** [More Information Needed]
|
| 25 |
+
- **License:** [More Information Needed]
|
| 26 |
+
- **Finetuned from model [optional]:** [More Information Needed]
|
| 27 |
+
|
| 28 |
+
### Model Sources [optional]
|
| 29 |
+
|
| 30 |
+
<!-- Provide the basic links for the model. -->
|
| 31 |
+
|
| 32 |
+
- **Repository:** [More Information Needed]
|
| 33 |
+
- **Paper [optional]:** [More Information Needed]
|
| 34 |
+
- **Demo [optional]:** [More Information Needed]
|
| 35 |
+
|
| 36 |
+
## Uses
|
| 37 |
+
|
| 38 |
+
<!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
|
| 39 |
+
|
| 40 |
+
### Direct Use
|
| 41 |
+
|
| 42 |
+
<!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
|
| 43 |
+
|
| 44 |
+
[More Information Needed]
|
| 45 |
+
|
| 46 |
+
### Downstream Use [optional]
|
| 47 |
+
|
| 48 |
+
<!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
|
| 49 |
+
|
| 50 |
+
[More Information Needed]
|
| 51 |
+
|
| 52 |
+
### Out-of-Scope Use
|
| 53 |
+
|
| 54 |
+
<!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
|
| 55 |
+
|
| 56 |
+
[More Information Needed]
|
| 57 |
+
|
| 58 |
+
## Bias, Risks, and Limitations
|
| 59 |
+
|
| 60 |
+
<!-- This section is meant to convey both technical and sociotechnical limitations. -->
|
| 61 |
+
|
| 62 |
+
[More Information Needed]
|
| 63 |
+
|
| 64 |
+
### Recommendations
|
| 65 |
+
|
| 66 |
+
<!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
|
| 67 |
+
|
| 68 |
+
Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
|
| 69 |
+
|
| 70 |
+
## How to Get Started with the Model
|
| 71 |
+
|
| 72 |
+
Use the code below to get started with the model.
|
| 73 |
+
|
| 74 |
+
[More Information Needed]
|
| 75 |
+
|
| 76 |
+
## Training Details
|
| 77 |
+
|
| 78 |
+
### Training Data
|
| 79 |
+
|
| 80 |
+
<!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
|
| 81 |
+
|
| 82 |
+
[More Information Needed]
|
| 83 |
+
|
| 84 |
+
### Training Procedure
|
| 85 |
+
|
| 86 |
+
<!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
|
| 87 |
+
|
| 88 |
+
#### Preprocessing [optional]
|
| 89 |
+
|
| 90 |
+
[More Information Needed]
|
| 91 |
+
|
| 92 |
+
|
| 93 |
+
#### Training Hyperparameters
|
| 94 |
+
|
| 95 |
+
- **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
|
| 96 |
+
|
| 97 |
+
#### Speeds, Sizes, Times [optional]
|
| 98 |
+
|
| 99 |
+
<!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
|
| 100 |
+
|
| 101 |
+
[More Information Needed]
|
| 102 |
+
|
| 103 |
+
## Evaluation
|
| 104 |
+
|
| 105 |
+
<!-- This section describes the evaluation protocols and provides the results. -->
|
| 106 |
+
|
| 107 |
+
### Testing Data, Factors & Metrics
|
| 108 |
+
|
| 109 |
+
#### Testing Data
|
| 110 |
+
|
| 111 |
+
<!-- This should link to a Dataset Card if possible. -->
|
| 112 |
+
|
| 113 |
+
[More Information Needed]
|
| 114 |
+
|
| 115 |
+
#### Factors
|
| 116 |
+
|
| 117 |
+
<!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
|
| 118 |
+
|
| 119 |
+
[More Information Needed]
|
| 120 |
+
|
| 121 |
+
#### Metrics
|
| 122 |
+
|
| 123 |
+
<!-- These are the evaluation metrics being used, ideally with a description of why. -->
|
| 124 |
+
|
| 125 |
+
[More Information Needed]
|
| 126 |
+
|
| 127 |
+
### Results
|
| 128 |
+
|
| 129 |
+
[More Information Needed]
|
| 130 |
+
|
| 131 |
+
#### Summary
|
| 132 |
+
|
| 133 |
+
|
| 134 |
+
|
| 135 |
+
## Model Examination [optional]
|
| 136 |
+
|
| 137 |
+
<!-- Relevant interpretability work for the model goes here -->
|
| 138 |
+
|
| 139 |
+
[More Information Needed]
|
| 140 |
+
|
| 141 |
+
## Environmental Impact
|
| 142 |
+
|
| 143 |
+
<!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
|
| 144 |
+
|
| 145 |
+
Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
|
| 146 |
+
|
| 147 |
+
- **Hardware Type:** [More Information Needed]
|
| 148 |
+
- **Hours used:** [More Information Needed]
|
| 149 |
+
- **Cloud Provider:** [More Information Needed]
|
| 150 |
+
- **Compute Region:** [More Information Needed]
|
| 151 |
+
- **Carbon Emitted:** [More Information Needed]
|
| 152 |
+
|
| 153 |
+
## Technical Specifications [optional]
|
| 154 |
+
|
| 155 |
+
### Model Architecture and Objective
|
| 156 |
+
|
| 157 |
+
[More Information Needed]
|
| 158 |
+
|
| 159 |
+
### Compute Infrastructure
|
| 160 |
+
|
| 161 |
+
[More Information Needed]
|
| 162 |
+
|
| 163 |
+
#### Hardware
|
| 164 |
+
|
| 165 |
+
[More Information Needed]
|
| 166 |
+
|
| 167 |
+
#### Software
|
| 168 |
+
|
| 169 |
+
[More Information Needed]
|
| 170 |
+
|
| 171 |
+
## Citation [optional]
|
| 172 |
+
|
| 173 |
+
<!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
|
| 174 |
+
|
| 175 |
+
**BibTeX:**
|
| 176 |
+
|
| 177 |
+
[More Information Needed]
|
| 178 |
+
|
| 179 |
+
**APA:**
|
| 180 |
+
|
| 181 |
+
[More Information Needed]
|
| 182 |
+
|
| 183 |
+
## Glossary [optional]
|
| 184 |
+
|
| 185 |
+
<!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
|
| 186 |
+
|
| 187 |
+
[More Information Needed]
|
| 188 |
+
|
| 189 |
+
## More Information [optional]
|
| 190 |
+
|
| 191 |
+
[More Information Needed]
|
| 192 |
+
|
| 193 |
+
## Model Card Authors [optional]
|
| 194 |
+
|
| 195 |
+
[More Information Needed]
|
| 196 |
+
|
| 197 |
+
## Model Card Contact
|
| 198 |
+
|
| 199 |
+
[More Information Needed]
|
| 200 |
+
### Framework versions
|
| 201 |
+
|
| 202 |
+
- PEFT 0.10.0
|
pels_phi2_qlora/adapter_config.json
ADDED
|
@@ -0,0 +1,33 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"alpha_pattern": {},
|
| 3 |
+
"auto_mapping": null,
|
| 4 |
+
"base_model_name_or_path": "microsoft/phi-2",
|
| 5 |
+
"bias": "none",
|
| 6 |
+
"fan_in_fan_out": false,
|
| 7 |
+
"inference_mode": true,
|
| 8 |
+
"init_lora_weights": true,
|
| 9 |
+
"layer_replication": null,
|
| 10 |
+
"layers_pattern": null,
|
| 11 |
+
"layers_to_transform": null,
|
| 12 |
+
"loftq_config": {},
|
| 13 |
+
"lora_alpha": 16,
|
| 14 |
+
"lora_dropout": 0.05,
|
| 15 |
+
"megatron_config": null,
|
| 16 |
+
"megatron_core": "megatron.core",
|
| 17 |
+
"modules_to_save": null,
|
| 18 |
+
"peft_type": "LORA",
|
| 19 |
+
"r": 8,
|
| 20 |
+
"rank_pattern": {},
|
| 21 |
+
"revision": null,
|
| 22 |
+
"target_modules": [
|
| 23 |
+
"k_proj",
|
| 24 |
+
"dense",
|
| 25 |
+
"q_proj",
|
| 26 |
+
"fc1",
|
| 27 |
+
"fc2",
|
| 28 |
+
"v_proj"
|
| 29 |
+
],
|
| 30 |
+
"task_type": "CAUSAL_LM",
|
| 31 |
+
"use_dora": false,
|
| 32 |
+
"use_rslora": false
|
| 33 |
+
}
|
pels_phi2_qlora/adapter_model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:e3c5bac14c96c04f69ade5544d398c98498d872cd1d1f811482fda5883d37de2
|
| 3 |
+
size 47235968
|
pels_phi2_qlora/added_tokens.json
ADDED
|
@@ -0,0 +1,40 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"\t\t": 50294,
|
| 3 |
+
"\t\t\t": 50293,
|
| 4 |
+
"\t\t\t\t": 50292,
|
| 5 |
+
"\t\t\t\t\t": 50291,
|
| 6 |
+
"\t\t\t\t\t\t": 50290,
|
| 7 |
+
"\t\t\t\t\t\t\t": 50289,
|
| 8 |
+
"\t\t\t\t\t\t\t\t": 50288,
|
| 9 |
+
"\t\t\t\t\t\t\t\t\t": 50287,
|
| 10 |
+
" ": 50286,
|
| 11 |
+
" ": 50285,
|
| 12 |
+
" ": 50284,
|
| 13 |
+
" ": 50283,
|
| 14 |
+
" ": 50282,
|
| 15 |
+
" ": 50281,
|
| 16 |
+
" ": 50280,
|
| 17 |
+
" ": 50279,
|
| 18 |
+
" ": 50278,
|
| 19 |
+
" ": 50277,
|
| 20 |
+
" ": 50276,
|
| 21 |
+
" ": 50275,
|
| 22 |
+
" ": 50274,
|
| 23 |
+
" ": 50273,
|
| 24 |
+
" ": 50272,
|
| 25 |
+
" ": 50271,
|
| 26 |
+
" ": 50270,
|
| 27 |
+
" ": 50269,
|
| 28 |
+
" ": 50268,
|
| 29 |
+
" ": 50267,
|
| 30 |
+
" ": 50266,
|
| 31 |
+
" ": 50265,
|
| 32 |
+
" ": 50264,
|
| 33 |
+
" ": 50263,
|
| 34 |
+
" ": 50262,
|
| 35 |
+
" ": 50261,
|
| 36 |
+
" ": 50260,
|
| 37 |
+
" ": 50259,
|
| 38 |
+
" ": 50258,
|
| 39 |
+
" ": 50257
|
| 40 |
+
}
|
pels_phi2_qlora/checkpoint-400/README.md
ADDED
|
@@ -0,0 +1,202 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
base_model: microsoft/phi-2
|
| 3 |
+
library_name: peft
|
| 4 |
+
---
|
| 5 |
+
|
| 6 |
+
# Model Card for Model ID
|
| 7 |
+
|
| 8 |
+
<!-- Provide a quick summary of what the model is/does. -->
|
| 9 |
+
|
| 10 |
+
|
| 11 |
+
|
| 12 |
+
## Model Details
|
| 13 |
+
|
| 14 |
+
### Model Description
|
| 15 |
+
|
| 16 |
+
<!-- Provide a longer summary of what this model is. -->
|
| 17 |
+
|
| 18 |
+
|
| 19 |
+
|
| 20 |
+
- **Developed by:** [More Information Needed]
|
| 21 |
+
- **Funded by [optional]:** [More Information Needed]
|
| 22 |
+
- **Shared by [optional]:** [More Information Needed]
|
| 23 |
+
- **Model type:** [More Information Needed]
|
| 24 |
+
- **Language(s) (NLP):** [More Information Needed]
|
| 25 |
+
- **License:** [More Information Needed]
|
| 26 |
+
- **Finetuned from model [optional]:** [More Information Needed]
|
| 27 |
+
|
| 28 |
+
### Model Sources [optional]
|
| 29 |
+
|
| 30 |
+
<!-- Provide the basic links for the model. -->
|
| 31 |
+
|
| 32 |
+
- **Repository:** [More Information Needed]
|
| 33 |
+
- **Paper [optional]:** [More Information Needed]
|
| 34 |
+
- **Demo [optional]:** [More Information Needed]
|
| 35 |
+
|
| 36 |
+
## Uses
|
| 37 |
+
|
| 38 |
+
<!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
|
| 39 |
+
|
| 40 |
+
### Direct Use
|
| 41 |
+
|
| 42 |
+
<!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
|
| 43 |
+
|
| 44 |
+
[More Information Needed]
|
| 45 |
+
|
| 46 |
+
### Downstream Use [optional]
|
| 47 |
+
|
| 48 |
+
<!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
|
| 49 |
+
|
| 50 |
+
[More Information Needed]
|
| 51 |
+
|
| 52 |
+
### Out-of-Scope Use
|
| 53 |
+
|
| 54 |
+
<!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
|
| 55 |
+
|
| 56 |
+
[More Information Needed]
|
| 57 |
+
|
| 58 |
+
## Bias, Risks, and Limitations
|
| 59 |
+
|
| 60 |
+
<!-- This section is meant to convey both technical and sociotechnical limitations. -->
|
| 61 |
+
|
| 62 |
+
[More Information Needed]
|
| 63 |
+
|
| 64 |
+
### Recommendations
|
| 65 |
+
|
| 66 |
+
<!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
|
| 67 |
+
|
| 68 |
+
Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
|
| 69 |
+
|
| 70 |
+
## How to Get Started with the Model
|
| 71 |
+
|
| 72 |
+
Use the code below to get started with the model.
|
| 73 |
+
|
| 74 |
+
[More Information Needed]
|
| 75 |
+
|
| 76 |
+
## Training Details
|
| 77 |
+
|
| 78 |
+
### Training Data
|
| 79 |
+
|
| 80 |
+
<!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
|
| 81 |
+
|
| 82 |
+
[More Information Needed]
|
| 83 |
+
|
| 84 |
+
### Training Procedure
|
| 85 |
+
|
| 86 |
+
<!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
|
| 87 |
+
|
| 88 |
+
#### Preprocessing [optional]
|
| 89 |
+
|
| 90 |
+
[More Information Needed]
|
| 91 |
+
|
| 92 |
+
|
| 93 |
+
#### Training Hyperparameters
|
| 94 |
+
|
| 95 |
+
- **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
|
| 96 |
+
|
| 97 |
+
#### Speeds, Sizes, Times [optional]
|
| 98 |
+
|
| 99 |
+
<!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
|
| 100 |
+
|
| 101 |
+
[More Information Needed]
|
| 102 |
+
|
| 103 |
+
## Evaluation
|
| 104 |
+
|
| 105 |
+
<!-- This section describes the evaluation protocols and provides the results. -->
|
| 106 |
+
|
| 107 |
+
### Testing Data, Factors & Metrics
|
| 108 |
+
|
| 109 |
+
#### Testing Data
|
| 110 |
+
|
| 111 |
+
<!-- This should link to a Dataset Card if possible. -->
|
| 112 |
+
|
| 113 |
+
[More Information Needed]
|
| 114 |
+
|
| 115 |
+
#### Factors
|
| 116 |
+
|
| 117 |
+
<!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
|
| 118 |
+
|
| 119 |
+
[More Information Needed]
|
| 120 |
+
|
| 121 |
+
#### Metrics
|
| 122 |
+
|
| 123 |
+
<!-- These are the evaluation metrics being used, ideally with a description of why. -->
|
| 124 |
+
|
| 125 |
+
[More Information Needed]
|
| 126 |
+
|
| 127 |
+
### Results
|
| 128 |
+
|
| 129 |
+
[More Information Needed]
|
| 130 |
+
|
| 131 |
+
#### Summary
|
| 132 |
+
|
| 133 |
+
|
| 134 |
+
|
| 135 |
+
## Model Examination [optional]
|
| 136 |
+
|
| 137 |
+
<!-- Relevant interpretability work for the model goes here -->
|
| 138 |
+
|
| 139 |
+
[More Information Needed]
|
| 140 |
+
|
| 141 |
+
## Environmental Impact
|
| 142 |
+
|
| 143 |
+
<!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
|
| 144 |
+
|
| 145 |
+
Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
|
| 146 |
+
|
| 147 |
+
- **Hardware Type:** [More Information Needed]
|
| 148 |
+
- **Hours used:** [More Information Needed]
|
| 149 |
+
- **Cloud Provider:** [More Information Needed]
|
| 150 |
+
- **Compute Region:** [More Information Needed]
|
| 151 |
+
- **Carbon Emitted:** [More Information Needed]
|
| 152 |
+
|
| 153 |
+
## Technical Specifications [optional]
|
| 154 |
+
|
| 155 |
+
### Model Architecture and Objective
|
| 156 |
+
|
| 157 |
+
[More Information Needed]
|
| 158 |
+
|
| 159 |
+
### Compute Infrastructure
|
| 160 |
+
|
| 161 |
+
[More Information Needed]
|
| 162 |
+
|
| 163 |
+
#### Hardware
|
| 164 |
+
|
| 165 |
+
[More Information Needed]
|
| 166 |
+
|
| 167 |
+
#### Software
|
| 168 |
+
|
| 169 |
+
[More Information Needed]
|
| 170 |
+
|
| 171 |
+
## Citation [optional]
|
| 172 |
+
|
| 173 |
+
<!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
|
| 174 |
+
|
| 175 |
+
**BibTeX:**
|
| 176 |
+
|
| 177 |
+
[More Information Needed]
|
| 178 |
+
|
| 179 |
+
**APA:**
|
| 180 |
+
|
| 181 |
+
[More Information Needed]
|
| 182 |
+
|
| 183 |
+
## Glossary [optional]
|
| 184 |
+
|
| 185 |
+
<!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
|
| 186 |
+
|
| 187 |
+
[More Information Needed]
|
| 188 |
+
|
| 189 |
+
## More Information [optional]
|
| 190 |
+
|
| 191 |
+
[More Information Needed]
|
| 192 |
+
|
| 193 |
+
## Model Card Authors [optional]
|
| 194 |
+
|
| 195 |
+
[More Information Needed]
|
| 196 |
+
|
| 197 |
+
## Model Card Contact
|
| 198 |
+
|
| 199 |
+
[More Information Needed]
|
| 200 |
+
### Framework versions
|
| 201 |
+
|
| 202 |
+
- PEFT 0.10.0
|
pels_phi2_qlora/checkpoint-400/adapter_config.json
ADDED
|
@@ -0,0 +1,33 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"alpha_pattern": {},
|
| 3 |
+
"auto_mapping": null,
|
| 4 |
+
"base_model_name_or_path": "microsoft/phi-2",
|
| 5 |
+
"bias": "none",
|
| 6 |
+
"fan_in_fan_out": false,
|
| 7 |
+
"inference_mode": true,
|
| 8 |
+
"init_lora_weights": true,
|
| 9 |
+
"layer_replication": null,
|
| 10 |
+
"layers_pattern": null,
|
| 11 |
+
"layers_to_transform": null,
|
| 12 |
+
"loftq_config": {},
|
| 13 |
+
"lora_alpha": 16,
|
| 14 |
+
"lora_dropout": 0.05,
|
| 15 |
+
"megatron_config": null,
|
| 16 |
+
"megatron_core": "megatron.core",
|
| 17 |
+
"modules_to_save": null,
|
| 18 |
+
"peft_type": "LORA",
|
| 19 |
+
"r": 8,
|
| 20 |
+
"rank_pattern": {},
|
| 21 |
+
"revision": null,
|
| 22 |
+
"target_modules": [
|
| 23 |
+
"k_proj",
|
| 24 |
+
"dense",
|
| 25 |
+
"q_proj",
|
| 26 |
+
"fc1",
|
| 27 |
+
"fc2",
|
| 28 |
+
"v_proj"
|
| 29 |
+
],
|
| 30 |
+
"task_type": "CAUSAL_LM",
|
| 31 |
+
"use_dora": false,
|
| 32 |
+
"use_rslora": false
|
| 33 |
+
}
|
pels_phi2_qlora/checkpoint-400/adapter_model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:85b115721d164fef66a9a2137c66280b908f7a0b657fcf91461264856a5b9688
|
| 3 |
+
size 47235968
|
pels_phi2_qlora/checkpoint-400/added_tokens.json
ADDED
|
@@ -0,0 +1,40 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"\t\t": 50294,
|
| 3 |
+
"\t\t\t": 50293,
|
| 4 |
+
"\t\t\t\t": 50292,
|
| 5 |
+
"\t\t\t\t\t": 50291,
|
| 6 |
+
"\t\t\t\t\t\t": 50290,
|
| 7 |
+
"\t\t\t\t\t\t\t": 50289,
|
| 8 |
+
"\t\t\t\t\t\t\t\t": 50288,
|
| 9 |
+
"\t\t\t\t\t\t\t\t\t": 50287,
|
| 10 |
+
" ": 50286,
|
| 11 |
+
" ": 50285,
|
| 12 |
+
" ": 50284,
|
| 13 |
+
" ": 50283,
|
| 14 |
+
" ": 50282,
|
| 15 |
+
" ": 50281,
|
| 16 |
+
" ": 50280,
|
| 17 |
+
" ": 50279,
|
| 18 |
+
" ": 50278,
|
| 19 |
+
" ": 50277,
|
| 20 |
+
" ": 50276,
|
| 21 |
+
" ": 50275,
|
| 22 |
+
" ": 50274,
|
| 23 |
+
" ": 50273,
|
| 24 |
+
" ": 50272,
|
| 25 |
+
" ": 50271,
|
| 26 |
+
" ": 50270,
|
| 27 |
+
" ": 50269,
|
| 28 |
+
" ": 50268,
|
| 29 |
+
" ": 50267,
|
| 30 |
+
" ": 50266,
|
| 31 |
+
" ": 50265,
|
| 32 |
+
" ": 50264,
|
| 33 |
+
" ": 50263,
|
| 34 |
+
" ": 50262,
|
| 35 |
+
" ": 50261,
|
| 36 |
+
" ": 50260,
|
| 37 |
+
" ": 50259,
|
| 38 |
+
" ": 50258,
|
| 39 |
+
" ": 50257
|
| 40 |
+
}
|
pels_phi2_qlora/checkpoint-400/merges.txt
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
pels_phi2_qlora/checkpoint-400/optimizer.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:98bd0f3ed730a9b0bd834268dd2b26e1491b42f1afaeabe546a27b54e474ad5a
|
| 3 |
+
size 24058644
|
pels_phi2_qlora/checkpoint-400/rng_state.pth
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:46aec81e2d2e70402f941d1568ab7a51a6e98b4769506a133ff4d4cb38bfdd72
|
| 3 |
+
size 14244
|
pels_phi2_qlora/checkpoint-400/scheduler.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:0825c00bcdc1810d5a1ee0d75db1465950c66bcf55eb26cbdcdc4ada4cfef52f
|
| 3 |
+
size 1064
|
pels_phi2_qlora/checkpoint-400/special_tokens_map.json
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"bos_token": {
|
| 3 |
+
"content": "<|endoftext|>",
|
| 4 |
+
"lstrip": false,
|
| 5 |
+
"normalized": false,
|
| 6 |
+
"rstrip": false,
|
| 7 |
+
"single_word": false
|
| 8 |
+
},
|
| 9 |
+
"eos_token": {
|
| 10 |
+
"content": "<|endoftext|>",
|
| 11 |
+
"lstrip": false,
|
| 12 |
+
"normalized": false,
|
| 13 |
+
"rstrip": false,
|
| 14 |
+
"single_word": false
|
| 15 |
+
},
|
| 16 |
+
"pad_token": "<|endoftext|>",
|
| 17 |
+
"unk_token": {
|
| 18 |
+
"content": "<|endoftext|>",
|
| 19 |
+
"lstrip": false,
|
| 20 |
+
"normalized": false,
|
| 21 |
+
"rstrip": false,
|
| 22 |
+
"single_word": false
|
| 23 |
+
}
|
| 24 |
+
}
|
pels_phi2_qlora/checkpoint-400/tokenizer.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
pels_phi2_qlora/checkpoint-400/tokenizer_config.json
ADDED
|
@@ -0,0 +1,325 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"add_prefix_space": false,
|
| 3 |
+
"added_tokens_decoder": {
|
| 4 |
+
"50256": {
|
| 5 |
+
"content": "<|endoftext|>",
|
| 6 |
+
"lstrip": false,
|
| 7 |
+
"normalized": false,
|
| 8 |
+
"rstrip": false,
|
| 9 |
+
"single_word": false,
|
| 10 |
+
"special": true
|
| 11 |
+
},
|
| 12 |
+
"50257": {
|
| 13 |
+
"content": " ",
|
| 14 |
+
"lstrip": false,
|
| 15 |
+
"normalized": true,
|
| 16 |
+
"rstrip": false,
|
| 17 |
+
"single_word": false,
|
| 18 |
+
"special": false
|
| 19 |
+
},
|
| 20 |
+
"50258": {
|
| 21 |
+
"content": " ",
|
| 22 |
+
"lstrip": false,
|
| 23 |
+
"normalized": true,
|
| 24 |
+
"rstrip": false,
|
| 25 |
+
"single_word": false,
|
| 26 |
+
"special": false
|
| 27 |
+
},
|
| 28 |
+
"50259": {
|
| 29 |
+
"content": " ",
|
| 30 |
+
"lstrip": false,
|
| 31 |
+
"normalized": true,
|
| 32 |
+
"rstrip": false,
|
| 33 |
+
"single_word": false,
|
| 34 |
+
"special": false
|
| 35 |
+
},
|
| 36 |
+
"50260": {
|
| 37 |
+
"content": " ",
|
| 38 |
+
"lstrip": false,
|
| 39 |
+
"normalized": true,
|
| 40 |
+
"rstrip": false,
|
| 41 |
+
"single_word": false,
|
| 42 |
+
"special": false
|
| 43 |
+
},
|
| 44 |
+
"50261": {
|
| 45 |
+
"content": " ",
|
| 46 |
+
"lstrip": false,
|
| 47 |
+
"normalized": true,
|
| 48 |
+
"rstrip": false,
|
| 49 |
+
"single_word": false,
|
| 50 |
+
"special": false
|
| 51 |
+
},
|
| 52 |
+
"50262": {
|
| 53 |
+
"content": " ",
|
| 54 |
+
"lstrip": false,
|
| 55 |
+
"normalized": true,
|
| 56 |
+
"rstrip": false,
|
| 57 |
+
"single_word": false,
|
| 58 |
+
"special": false
|
| 59 |
+
},
|
| 60 |
+
"50263": {
|
| 61 |
+
"content": " ",
|
| 62 |
+
"lstrip": false,
|
| 63 |
+
"normalized": true,
|
| 64 |
+
"rstrip": false,
|
| 65 |
+
"single_word": false,
|
| 66 |
+
"special": false
|
| 67 |
+
},
|
| 68 |
+
"50264": {
|
| 69 |
+
"content": " ",
|
| 70 |
+
"lstrip": false,
|
| 71 |
+
"normalized": true,
|
| 72 |
+
"rstrip": false,
|
| 73 |
+
"single_word": false,
|
| 74 |
+
"special": false
|
| 75 |
+
},
|
| 76 |
+
"50265": {
|
| 77 |
+
"content": " ",
|
| 78 |
+
"lstrip": false,
|
| 79 |
+
"normalized": true,
|
| 80 |
+
"rstrip": false,
|
| 81 |
+
"single_word": false,
|
| 82 |
+
"special": false
|
| 83 |
+
},
|
| 84 |
+
"50266": {
|
| 85 |
+
"content": " ",
|
| 86 |
+
"lstrip": false,
|
| 87 |
+
"normalized": true,
|
| 88 |
+
"rstrip": false,
|
| 89 |
+
"single_word": false,
|
| 90 |
+
"special": false
|
| 91 |
+
},
|
| 92 |
+
"50267": {
|
| 93 |
+
"content": " ",
|
| 94 |
+
"lstrip": false,
|
| 95 |
+
"normalized": true,
|
| 96 |
+
"rstrip": false,
|
| 97 |
+
"single_word": false,
|
| 98 |
+
"special": false
|
| 99 |
+
},
|
| 100 |
+
"50268": {
|
| 101 |
+
"content": " ",
|
| 102 |
+
"lstrip": false,
|
| 103 |
+
"normalized": true,
|
| 104 |
+
"rstrip": false,
|
| 105 |
+
"single_word": false,
|
| 106 |
+
"special": false
|
| 107 |
+
},
|
| 108 |
+
"50269": {
|
| 109 |
+
"content": " ",
|
| 110 |
+
"lstrip": false,
|
| 111 |
+
"normalized": true,
|
| 112 |
+
"rstrip": false,
|
| 113 |
+
"single_word": false,
|
| 114 |
+
"special": false
|
| 115 |
+
},
|
| 116 |
+
"50270": {
|
| 117 |
+
"content": " ",
|
| 118 |
+
"lstrip": false,
|
| 119 |
+
"normalized": true,
|
| 120 |
+
"rstrip": false,
|
| 121 |
+
"single_word": false,
|
| 122 |
+
"special": false
|
| 123 |
+
},
|
| 124 |
+
"50271": {
|
| 125 |
+
"content": " ",
|
| 126 |
+
"lstrip": false,
|
| 127 |
+
"normalized": true,
|
| 128 |
+
"rstrip": false,
|
| 129 |
+
"single_word": false,
|
| 130 |
+
"special": false
|
| 131 |
+
},
|
| 132 |
+
"50272": {
|
| 133 |
+
"content": " ",
|
| 134 |
+
"lstrip": false,
|
| 135 |
+
"normalized": true,
|
| 136 |
+
"rstrip": false,
|
| 137 |
+
"single_word": false,
|
| 138 |
+
"special": false
|
| 139 |
+
},
|
| 140 |
+
"50273": {
|
| 141 |
+
"content": " ",
|
| 142 |
+
"lstrip": false,
|
| 143 |
+
"normalized": true,
|
| 144 |
+
"rstrip": false,
|
| 145 |
+
"single_word": false,
|
| 146 |
+
"special": false
|
| 147 |
+
},
|
| 148 |
+
"50274": {
|
| 149 |
+
"content": " ",
|
| 150 |
+
"lstrip": false,
|
| 151 |
+
"normalized": true,
|
| 152 |
+
"rstrip": false,
|
| 153 |
+
"single_word": false,
|
| 154 |
+
"special": false
|
| 155 |
+
},
|
| 156 |
+
"50275": {
|
| 157 |
+
"content": " ",
|
| 158 |
+
"lstrip": false,
|
| 159 |
+
"normalized": true,
|
| 160 |
+
"rstrip": false,
|
| 161 |
+
"single_word": false,
|
| 162 |
+
"special": false
|
| 163 |
+
},
|
| 164 |
+
"50276": {
|
| 165 |
+
"content": " ",
|
| 166 |
+
"lstrip": false,
|
| 167 |
+
"normalized": true,
|
| 168 |
+
"rstrip": false,
|
| 169 |
+
"single_word": false,
|
| 170 |
+
"special": false
|
| 171 |
+
},
|
| 172 |
+
"50277": {
|
| 173 |
+
"content": " ",
|
| 174 |
+
"lstrip": false,
|
| 175 |
+
"normalized": true,
|
| 176 |
+
"rstrip": false,
|
| 177 |
+
"single_word": false,
|
| 178 |
+
"special": false
|
| 179 |
+
},
|
| 180 |
+
"50278": {
|
| 181 |
+
"content": " ",
|
| 182 |
+
"lstrip": false,
|
| 183 |
+
"normalized": true,
|
| 184 |
+
"rstrip": false,
|
| 185 |
+
"single_word": false,
|
| 186 |
+
"special": false
|
| 187 |
+
},
|
| 188 |
+
"50279": {
|
| 189 |
+
"content": " ",
|
| 190 |
+
"lstrip": false,
|
| 191 |
+
"normalized": true,
|
| 192 |
+
"rstrip": false,
|
| 193 |
+
"single_word": false,
|
| 194 |
+
"special": false
|
| 195 |
+
},
|
| 196 |
+
"50280": {
|
| 197 |
+
"content": " ",
|
| 198 |
+
"lstrip": false,
|
| 199 |
+
"normalized": true,
|
| 200 |
+
"rstrip": false,
|
| 201 |
+
"single_word": false,
|
| 202 |
+
"special": false
|
| 203 |
+
},
|
| 204 |
+
"50281": {
|
| 205 |
+
"content": " ",
|
| 206 |
+
"lstrip": false,
|
| 207 |
+
"normalized": true,
|
| 208 |
+
"rstrip": false,
|
| 209 |
+
"single_word": false,
|
| 210 |
+
"special": false
|
| 211 |
+
},
|
| 212 |
+
"50282": {
|
| 213 |
+
"content": " ",
|
| 214 |
+
"lstrip": false,
|
| 215 |
+
"normalized": true,
|
| 216 |
+
"rstrip": false,
|
| 217 |
+
"single_word": false,
|
| 218 |
+
"special": false
|
| 219 |
+
},
|
| 220 |
+
"50283": {
|
| 221 |
+
"content": " ",
|
| 222 |
+
"lstrip": false,
|
| 223 |
+
"normalized": true,
|
| 224 |
+
"rstrip": false,
|
| 225 |
+
"single_word": false,
|
| 226 |
+
"special": false
|
| 227 |
+
},
|
| 228 |
+
"50284": {
|
| 229 |
+
"content": " ",
|
| 230 |
+
"lstrip": false,
|
| 231 |
+
"normalized": true,
|
| 232 |
+
"rstrip": false,
|
| 233 |
+
"single_word": false,
|
| 234 |
+
"special": false
|
| 235 |
+
},
|
| 236 |
+
"50285": {
|
| 237 |
+
"content": " ",
|
| 238 |
+
"lstrip": false,
|
| 239 |
+
"normalized": true,
|
| 240 |
+
"rstrip": false,
|
| 241 |
+
"single_word": false,
|
| 242 |
+
"special": false
|
| 243 |
+
},
|
| 244 |
+
"50286": {
|
| 245 |
+
"content": " ",
|
| 246 |
+
"lstrip": false,
|
| 247 |
+
"normalized": true,
|
| 248 |
+
"rstrip": false,
|
| 249 |
+
"single_word": false,
|
| 250 |
+
"special": false
|
| 251 |
+
},
|
| 252 |
+
"50287": {
|
| 253 |
+
"content": "\t\t\t\t\t\t\t\t\t",
|
| 254 |
+
"lstrip": false,
|
| 255 |
+
"normalized": true,
|
| 256 |
+
"rstrip": false,
|
| 257 |
+
"single_word": false,
|
| 258 |
+
"special": false
|
| 259 |
+
},
|
| 260 |
+
"50288": {
|
| 261 |
+
"content": "\t\t\t\t\t\t\t\t",
|
| 262 |
+
"lstrip": false,
|
| 263 |
+
"normalized": true,
|
| 264 |
+
"rstrip": false,
|
| 265 |
+
"single_word": false,
|
| 266 |
+
"special": false
|
| 267 |
+
},
|
| 268 |
+
"50289": {
|
| 269 |
+
"content": "\t\t\t\t\t\t\t",
|
| 270 |
+
"lstrip": false,
|
| 271 |
+
"normalized": true,
|
| 272 |
+
"rstrip": false,
|
| 273 |
+
"single_word": false,
|
| 274 |
+
"special": false
|
| 275 |
+
},
|
| 276 |
+
"50290": {
|
| 277 |
+
"content": "\t\t\t\t\t\t",
|
| 278 |
+
"lstrip": false,
|
| 279 |
+
"normalized": true,
|
| 280 |
+
"rstrip": false,
|
| 281 |
+
"single_word": false,
|
| 282 |
+
"special": false
|
| 283 |
+
},
|
| 284 |
+
"50291": {
|
| 285 |
+
"content": "\t\t\t\t\t",
|
| 286 |
+
"lstrip": false,
|
| 287 |
+
"normalized": true,
|
| 288 |
+
"rstrip": false,
|
| 289 |
+
"single_word": false,
|
| 290 |
+
"special": false
|
| 291 |
+
},
|
| 292 |
+
"50292": {
|
| 293 |
+
"content": "\t\t\t\t",
|
| 294 |
+
"lstrip": false,
|
| 295 |
+
"normalized": true,
|
| 296 |
+
"rstrip": false,
|
| 297 |
+
"single_word": false,
|
| 298 |
+
"special": false
|
| 299 |
+
},
|
| 300 |
+
"50293": {
|
| 301 |
+
"content": "\t\t\t",
|
| 302 |
+
"lstrip": false,
|
| 303 |
+
"normalized": true,
|
| 304 |
+
"rstrip": false,
|
| 305 |
+
"single_word": false,
|
| 306 |
+
"special": false
|
| 307 |
+
},
|
| 308 |
+
"50294": {
|
| 309 |
+
"content": "\t\t",
|
| 310 |
+
"lstrip": false,
|
| 311 |
+
"normalized": true,
|
| 312 |
+
"rstrip": false,
|
| 313 |
+
"single_word": false,
|
| 314 |
+
"special": false
|
| 315 |
+
}
|
| 316 |
+
},
|
| 317 |
+
"bos_token": "<|endoftext|>",
|
| 318 |
+
"clean_up_tokenization_spaces": true,
|
| 319 |
+
"eos_token": "<|endoftext|>",
|
| 320 |
+
"model_max_length": 2048,
|
| 321 |
+
"pad_token": "<|endoftext|>",
|
| 322 |
+
"return_token_type_ids": false,
|
| 323 |
+
"tokenizer_class": "CodeGenTokenizer",
|
| 324 |
+
"unk_token": "<|endoftext|>"
|
| 325 |
+
}
|
pels_phi2_qlora/checkpoint-400/trainer_state.json
ADDED
|
@@ -0,0 +1,165 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"best_metric": 0.25696343183517456,
|
| 3 |
+
"best_model_checkpoint": "./pels_phi2_qlora\\checkpoint-400",
|
| 4 |
+
"epoch": 2.388059701492537,
|
| 5 |
+
"eval_steps": 100,
|
| 6 |
+
"global_step": 400,
|
| 7 |
+
"is_hyper_param_search": false,
|
| 8 |
+
"is_local_process_zero": true,
|
| 9 |
+
"is_world_process_zero": true,
|
| 10 |
+
"log_history": [
|
| 11 |
+
{
|
| 12 |
+
"epoch": 0.14925373134328357,
|
| 13 |
+
"grad_norm": 1.4428153038024902,
|
| 14 |
+
"learning_rate": 0.00019230769230769233,
|
| 15 |
+
"loss": 2.6364,
|
| 16 |
+
"step": 25
|
| 17 |
+
},
|
| 18 |
+
{
|
| 19 |
+
"epoch": 0.29850746268656714,
|
| 20 |
+
"grad_norm": 0.8800177574157715,
|
| 21 |
+
"learning_rate": 0.00019874283308955057,
|
| 22 |
+
"loss": 1.0969,
|
| 23 |
+
"step": 50
|
| 24 |
+
},
|
| 25 |
+
{
|
| 26 |
+
"epoch": 0.44776119402985076,
|
| 27 |
+
"grad_norm": 0.8356912732124329,
|
| 28 |
+
"learning_rate": 0.0001947944062577507,
|
| 29 |
+
"loss": 0.6956,
|
| 30 |
+
"step": 75
|
| 31 |
+
},
|
| 32 |
+
{
|
| 33 |
+
"epoch": 0.5970149253731343,
|
| 34 |
+
"grad_norm": 0.660551905632019,
|
| 35 |
+
"learning_rate": 0.0001882602351338137,
|
| 36 |
+
"loss": 0.4576,
|
| 37 |
+
"step": 100
|
| 38 |
+
},
|
| 39 |
+
{
|
| 40 |
+
"epoch": 0.5970149253731343,
|
| 41 |
+
"eval_loss": 0.3502635955810547,
|
| 42 |
+
"eval_runtime": 116.3908,
|
| 43 |
+
"eval_samples_per_second": 1.435,
|
| 44 |
+
"eval_steps_per_second": 1.435,
|
| 45 |
+
"step": 100
|
| 46 |
+
},
|
| 47 |
+
{
|
| 48 |
+
"epoch": 0.746268656716418,
|
| 49 |
+
"grad_norm": 0.6952309608459473,
|
| 50 |
+
"learning_rate": 0.00017931855487268782,
|
| 51 |
+
"loss": 0.3221,
|
| 52 |
+
"step": 125
|
| 53 |
+
},
|
| 54 |
+
{
|
| 55 |
+
"epoch": 0.8955223880597015,
|
| 56 |
+
"grad_norm": 0.43301114439964294,
|
| 57 |
+
"learning_rate": 0.00016821327120267567,
|
| 58 |
+
"loss": 0.4255,
|
| 59 |
+
"step": 150
|
| 60 |
+
},
|
| 61 |
+
{
|
| 62 |
+
"epoch": 1.044776119402985,
|
| 63 |
+
"grad_norm": 0.3896200954914093,
|
| 64 |
+
"learning_rate": 0.00015524730731298134,
|
| 65 |
+
"loss": 0.2244,
|
| 66 |
+
"step": 175
|
| 67 |
+
},
|
| 68 |
+
{
|
| 69 |
+
"epoch": 1.1940298507462686,
|
| 70 |
+
"grad_norm": 0.33569812774658203,
|
| 71 |
+
"learning_rate": 0.00014077434089877037,
|
| 72 |
+
"loss": 0.3321,
|
| 73 |
+
"step": 200
|
| 74 |
+
},
|
| 75 |
+
{
|
| 76 |
+
"epoch": 1.1940298507462686,
|
| 77 |
+
"eval_loss": 0.27684858441352844,
|
| 78 |
+
"eval_runtime": 115.5816,
|
| 79 |
+
"eval_samples_per_second": 1.445,
|
| 80 |
+
"eval_steps_per_second": 1.445,
|
| 81 |
+
"step": 200
|
| 82 |
+
},
|
| 83 |
+
{
|
| 84 |
+
"epoch": 1.3432835820895521,
|
| 85 |
+
"grad_norm": 0.41767171025276184,
|
| 86 |
+
"learning_rate": 0.00012518915675561483,
|
| 87 |
+
"loss": 0.2908,
|
| 88 |
+
"step": 225
|
| 89 |
+
},
|
| 90 |
+
{
|
| 91 |
+
"epoch": 1.4925373134328357,
|
| 92 |
+
"grad_norm": 0.38486623764038086,
|
| 93 |
+
"learning_rate": 0.00010891687807939707,
|
| 94 |
+
"loss": 0.2308,
|
| 95 |
+
"step": 250
|
| 96 |
+
},
|
| 97 |
+
{
|
| 98 |
+
"epoch": 1.6417910447761193,
|
| 99 |
+
"grad_norm": 0.33858075737953186,
|
| 100 |
+
"learning_rate": 9.24013702137397e-05,
|
| 101 |
+
"loss": 0.3402,
|
| 102 |
+
"step": 275
|
| 103 |
+
},
|
| 104 |
+
{
|
| 105 |
+
"epoch": 1.7910447761194028,
|
| 106 |
+
"grad_norm": 0.37202343344688416,
|
| 107 |
+
"learning_rate": 7.6093133160502e-05,
|
| 108 |
+
"loss": 0.2394,
|
| 109 |
+
"step": 300
|
| 110 |
+
},
|
| 111 |
+
{
|
| 112 |
+
"epoch": 1.7910447761194028,
|
| 113 |
+
"eval_loss": 0.2645300626754761,
|
| 114 |
+
"eval_runtime": 117.2223,
|
| 115 |
+
"eval_samples_per_second": 1.425,
|
| 116 |
+
"eval_steps_per_second": 1.425,
|
| 117 |
+
"step": 300
|
| 118 |
+
},
|
| 119 |
+
{
|
| 120 |
+
"epoch": 1.9402985074626866,
|
| 121 |
+
"grad_norm": 0.2712254822254181,
|
| 122 |
+
"learning_rate": 6.0437013114095195e-05,
|
| 123 |
+
"loss": 0.323,
|
| 124 |
+
"step": 325
|
| 125 |
+
},
|
| 126 |
+
{
|
| 127 |
+
"epoch": 2.08955223880597,
|
| 128 |
+
"grad_norm": 0.3418550193309784,
|
| 129 |
+
"learning_rate": 4.58600682169262e-05,
|
| 130 |
+
"loss": 0.2754,
|
| 131 |
+
"step": 350
|
| 132 |
+
},
|
| 133 |
+
{
|
| 134 |
+
"epoch": 2.2388059701492535,
|
| 135 |
+
"grad_norm": 0.3196725845336914,
|
| 136 |
+
"learning_rate": 3.275991952653054e-05,
|
| 137 |
+
"loss": 0.2195,
|
| 138 |
+
"step": 375
|
| 139 |
+
},
|
| 140 |
+
{
|
| 141 |
+
"epoch": 2.388059701492537,
|
| 142 |
+
"grad_norm": 0.42392605543136597,
|
| 143 |
+
"learning_rate": 2.149390494964323e-05,
|
| 144 |
+
"loss": 0.3365,
|
| 145 |
+
"step": 400
|
| 146 |
+
},
|
| 147 |
+
{
|
| 148 |
+
"epoch": 2.388059701492537,
|
| 149 |
+
"eval_loss": 0.25696343183517456,
|
| 150 |
+
"eval_runtime": 116.1779,
|
| 151 |
+
"eval_samples_per_second": 1.437,
|
| 152 |
+
"eval_steps_per_second": 1.437,
|
| 153 |
+
"step": 400
|
| 154 |
+
}
|
| 155 |
+
],
|
| 156 |
+
"logging_steps": 25,
|
| 157 |
+
"max_steps": 501,
|
| 158 |
+
"num_input_tokens_seen": 0,
|
| 159 |
+
"num_train_epochs": 3,
|
| 160 |
+
"save_steps": 100,
|
| 161 |
+
"total_flos": 2.592156608713728e+16,
|
| 162 |
+
"train_batch_size": 1,
|
| 163 |
+
"trial_name": null,
|
| 164 |
+
"trial_params": null
|
| 165 |
+
}
|
pels_phi2_qlora/checkpoint-400/training_args.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:ffd1259f93baa6dbe0764efae1c295a80e7889d53f7c7bfc7c725d52eabc931f
|
| 3 |
+
size 4984
|
pels_phi2_qlora/checkpoint-400/vocab.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
pels_phi2_qlora/checkpoint-500/README.md
ADDED
|
@@ -0,0 +1,202 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
base_model: microsoft/phi-2
|
| 3 |
+
library_name: peft
|
| 4 |
+
---
|
| 5 |
+
|
| 6 |
+
# Model Card for Model ID
|
| 7 |
+
|
| 8 |
+
<!-- Provide a quick summary of what the model is/does. -->
|
| 9 |
+
|
| 10 |
+
|
| 11 |
+
|
| 12 |
+
## Model Details
|
| 13 |
+
|
| 14 |
+
### Model Description
|
| 15 |
+
|
| 16 |
+
<!-- Provide a longer summary of what this model is. -->
|
| 17 |
+
|
| 18 |
+
|
| 19 |
+
|
| 20 |
+
- **Developed by:** [More Information Needed]
|
| 21 |
+
- **Funded by [optional]:** [More Information Needed]
|
| 22 |
+
- **Shared by [optional]:** [More Information Needed]
|
| 23 |
+
- **Model type:** [More Information Needed]
|
| 24 |
+
- **Language(s) (NLP):** [More Information Needed]
|
| 25 |
+
- **License:** [More Information Needed]
|
| 26 |
+
- **Finetuned from model [optional]:** [More Information Needed]
|
| 27 |
+
|
| 28 |
+
### Model Sources [optional]
|
| 29 |
+
|
| 30 |
+
<!-- Provide the basic links for the model. -->
|
| 31 |
+
|
| 32 |
+
- **Repository:** [More Information Needed]
|
| 33 |
+
- **Paper [optional]:** [More Information Needed]
|
| 34 |
+
- **Demo [optional]:** [More Information Needed]
|
| 35 |
+
|
| 36 |
+
## Uses
|
| 37 |
+
|
| 38 |
+
<!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
|
| 39 |
+
|
| 40 |
+
### Direct Use
|
| 41 |
+
|
| 42 |
+
<!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
|
| 43 |
+
|
| 44 |
+
[More Information Needed]
|
| 45 |
+
|
| 46 |
+
### Downstream Use [optional]
|
| 47 |
+
|
| 48 |
+
<!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
|
| 49 |
+
|
| 50 |
+
[More Information Needed]
|
| 51 |
+
|
| 52 |
+
### Out-of-Scope Use
|
| 53 |
+
|
| 54 |
+
<!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
|
| 55 |
+
|
| 56 |
+
[More Information Needed]
|
| 57 |
+
|
| 58 |
+
## Bias, Risks, and Limitations
|
| 59 |
+
|
| 60 |
+
<!-- This section is meant to convey both technical and sociotechnical limitations. -->
|
| 61 |
+
|
| 62 |
+
[More Information Needed]
|
| 63 |
+
|
| 64 |
+
### Recommendations
|
| 65 |
+
|
| 66 |
+
<!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
|
| 67 |
+
|
| 68 |
+
Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
|
| 69 |
+
|
| 70 |
+
## How to Get Started with the Model
|
| 71 |
+
|
| 72 |
+
Use the code below to get started with the model.
|
| 73 |
+
|
| 74 |
+
[More Information Needed]
|
| 75 |
+
|
| 76 |
+
## Training Details
|
| 77 |
+
|
| 78 |
+
### Training Data
|
| 79 |
+
|
| 80 |
+
<!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
|
| 81 |
+
|
| 82 |
+
[More Information Needed]
|
| 83 |
+
|
| 84 |
+
### Training Procedure
|
| 85 |
+
|
| 86 |
+
<!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
|
| 87 |
+
|
| 88 |
+
#### Preprocessing [optional]
|
| 89 |
+
|
| 90 |
+
[More Information Needed]
|
| 91 |
+
|
| 92 |
+
|
| 93 |
+
#### Training Hyperparameters
|
| 94 |
+
|
| 95 |
+
- **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
|
| 96 |
+
|
| 97 |
+
#### Speeds, Sizes, Times [optional]
|
| 98 |
+
|
| 99 |
+
<!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
|
| 100 |
+
|
| 101 |
+
[More Information Needed]
|
| 102 |
+
|
| 103 |
+
## Evaluation
|
| 104 |
+
|
| 105 |
+
<!-- This section describes the evaluation protocols and provides the results. -->
|
| 106 |
+
|
| 107 |
+
### Testing Data, Factors & Metrics
|
| 108 |
+
|
| 109 |
+
#### Testing Data
|
| 110 |
+
|
| 111 |
+
<!-- This should link to a Dataset Card if possible. -->
|
| 112 |
+
|
| 113 |
+
[More Information Needed]
|
| 114 |
+
|
| 115 |
+
#### Factors
|
| 116 |
+
|
| 117 |
+
<!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
|
| 118 |
+
|
| 119 |
+
[More Information Needed]
|
| 120 |
+
|
| 121 |
+
#### Metrics
|
| 122 |
+
|
| 123 |
+
<!-- These are the evaluation metrics being used, ideally with a description of why. -->
|
| 124 |
+
|
| 125 |
+
[More Information Needed]
|
| 126 |
+
|
| 127 |
+
### Results
|
| 128 |
+
|
| 129 |
+
[More Information Needed]
|
| 130 |
+
|
| 131 |
+
#### Summary
|
| 132 |
+
|
| 133 |
+
|
| 134 |
+
|
| 135 |
+
## Model Examination [optional]
|
| 136 |
+
|
| 137 |
+
<!-- Relevant interpretability work for the model goes here -->
|
| 138 |
+
|
| 139 |
+
[More Information Needed]
|
| 140 |
+
|
| 141 |
+
## Environmental Impact
|
| 142 |
+
|
| 143 |
+
<!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
|
| 144 |
+
|
| 145 |
+
Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
|
| 146 |
+
|
| 147 |
+
- **Hardware Type:** [More Information Needed]
|
| 148 |
+
- **Hours used:** [More Information Needed]
|
| 149 |
+
- **Cloud Provider:** [More Information Needed]
|
| 150 |
+
- **Compute Region:** [More Information Needed]
|
| 151 |
+
- **Carbon Emitted:** [More Information Needed]
|
| 152 |
+
|
| 153 |
+
## Technical Specifications [optional]
|
| 154 |
+
|
| 155 |
+
### Model Architecture and Objective
|
| 156 |
+
|
| 157 |
+
[More Information Needed]
|
| 158 |
+
|
| 159 |
+
### Compute Infrastructure
|
| 160 |
+
|
| 161 |
+
[More Information Needed]
|
| 162 |
+
|
| 163 |
+
#### Hardware
|
| 164 |
+
|
| 165 |
+
[More Information Needed]
|
| 166 |
+
|
| 167 |
+
#### Software
|
| 168 |
+
|
| 169 |
+
[More Information Needed]
|
| 170 |
+
|
| 171 |
+
## Citation [optional]
|
| 172 |
+
|
| 173 |
+
<!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
|
| 174 |
+
|
| 175 |
+
**BibTeX:**
|
| 176 |
+
|
| 177 |
+
[More Information Needed]
|
| 178 |
+
|
| 179 |
+
**APA:**
|
| 180 |
+
|
| 181 |
+
[More Information Needed]
|
| 182 |
+
|
| 183 |
+
## Glossary [optional]
|
| 184 |
+
|
| 185 |
+
<!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
|
| 186 |
+
|
| 187 |
+
[More Information Needed]
|
| 188 |
+
|
| 189 |
+
## More Information [optional]
|
| 190 |
+
|
| 191 |
+
[More Information Needed]
|
| 192 |
+
|
| 193 |
+
## Model Card Authors [optional]
|
| 194 |
+
|
| 195 |
+
[More Information Needed]
|
| 196 |
+
|
| 197 |
+
## Model Card Contact
|
| 198 |
+
|
| 199 |
+
[More Information Needed]
|
| 200 |
+
### Framework versions
|
| 201 |
+
|
| 202 |
+
- PEFT 0.10.0
|
pels_phi2_qlora/checkpoint-500/adapter_config.json
ADDED
|
@@ -0,0 +1,33 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"alpha_pattern": {},
|
| 3 |
+
"auto_mapping": null,
|
| 4 |
+
"base_model_name_or_path": "microsoft/phi-2",
|
| 5 |
+
"bias": "none",
|
| 6 |
+
"fan_in_fan_out": false,
|
| 7 |
+
"inference_mode": true,
|
| 8 |
+
"init_lora_weights": true,
|
| 9 |
+
"layer_replication": null,
|
| 10 |
+
"layers_pattern": null,
|
| 11 |
+
"layers_to_transform": null,
|
| 12 |
+
"loftq_config": {},
|
| 13 |
+
"lora_alpha": 16,
|
| 14 |
+
"lora_dropout": 0.05,
|
| 15 |
+
"megatron_config": null,
|
| 16 |
+
"megatron_core": "megatron.core",
|
| 17 |
+
"modules_to_save": null,
|
| 18 |
+
"peft_type": "LORA",
|
| 19 |
+
"r": 8,
|
| 20 |
+
"rank_pattern": {},
|
| 21 |
+
"revision": null,
|
| 22 |
+
"target_modules": [
|
| 23 |
+
"k_proj",
|
| 24 |
+
"dense",
|
| 25 |
+
"q_proj",
|
| 26 |
+
"fc1",
|
| 27 |
+
"fc2",
|
| 28 |
+
"v_proj"
|
| 29 |
+
],
|
| 30 |
+
"task_type": "CAUSAL_LM",
|
| 31 |
+
"use_dora": false,
|
| 32 |
+
"use_rslora": false
|
| 33 |
+
}
|
pels_phi2_qlora/checkpoint-500/adapter_model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:e3c5bac14c96c04f69ade5544d398c98498d872cd1d1f811482fda5883d37de2
|
| 3 |
+
size 47235968
|
pels_phi2_qlora/checkpoint-500/added_tokens.json
ADDED
|
@@ -0,0 +1,40 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"\t\t": 50294,
|
| 3 |
+
"\t\t\t": 50293,
|
| 4 |
+
"\t\t\t\t": 50292,
|
| 5 |
+
"\t\t\t\t\t": 50291,
|
| 6 |
+
"\t\t\t\t\t\t": 50290,
|
| 7 |
+
"\t\t\t\t\t\t\t": 50289,
|
| 8 |
+
"\t\t\t\t\t\t\t\t": 50288,
|
| 9 |
+
"\t\t\t\t\t\t\t\t\t": 50287,
|
| 10 |
+
" ": 50286,
|
| 11 |
+
" ": 50285,
|
| 12 |
+
" ": 50284,
|
| 13 |
+
" ": 50283,
|
| 14 |
+
" ": 50282,
|
| 15 |
+
" ": 50281,
|
| 16 |
+
" ": 50280,
|
| 17 |
+
" ": 50279,
|
| 18 |
+
" ": 50278,
|
| 19 |
+
" ": 50277,
|
| 20 |
+
" ": 50276,
|
| 21 |
+
" ": 50275,
|
| 22 |
+
" ": 50274,
|
| 23 |
+
" ": 50273,
|
| 24 |
+
" ": 50272,
|
| 25 |
+
" ": 50271,
|
| 26 |
+
" ": 50270,
|
| 27 |
+
" ": 50269,
|
| 28 |
+
" ": 50268,
|
| 29 |
+
" ": 50267,
|
| 30 |
+
" ": 50266,
|
| 31 |
+
" ": 50265,
|
| 32 |
+
" ": 50264,
|
| 33 |
+
" ": 50263,
|
| 34 |
+
" ": 50262,
|
| 35 |
+
" ": 50261,
|
| 36 |
+
" ": 50260,
|
| 37 |
+
" ": 50259,
|
| 38 |
+
" ": 50258,
|
| 39 |
+
" ": 50257
|
| 40 |
+
}
|
pels_phi2_qlora/checkpoint-500/merges.txt
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
pels_phi2_qlora/checkpoint-500/optimizer.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:44abb7aa90450feade39092900d55f7b44f1294b0bf3cc86205de1a5ac0fee75
|
| 3 |
+
size 24058644
|
pels_phi2_qlora/checkpoint-500/rng_state.pth
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:565e63cf9fe3a03ddd7f26710dec70af1972124065f46995a6d2e9e5dc292aa6
|
| 3 |
+
size 14244
|
pels_phi2_qlora/checkpoint-500/scheduler.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:93230455c8ab1f3da9d11c9c3144568c3d2acfa4ff4e8b4b8a55d3910f342d71
|
| 3 |
+
size 1064
|
pels_phi2_qlora/checkpoint-500/special_tokens_map.json
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"bos_token": {
|
| 3 |
+
"content": "<|endoftext|>",
|
| 4 |
+
"lstrip": false,
|
| 5 |
+
"normalized": false,
|
| 6 |
+
"rstrip": false,
|
| 7 |
+
"single_word": false
|
| 8 |
+
},
|
| 9 |
+
"eos_token": {
|
| 10 |
+
"content": "<|endoftext|>",
|
| 11 |
+
"lstrip": false,
|
| 12 |
+
"normalized": false,
|
| 13 |
+
"rstrip": false,
|
| 14 |
+
"single_word": false
|
| 15 |
+
},
|
| 16 |
+
"pad_token": "<|endoftext|>",
|
| 17 |
+
"unk_token": {
|
| 18 |
+
"content": "<|endoftext|>",
|
| 19 |
+
"lstrip": false,
|
| 20 |
+
"normalized": false,
|
| 21 |
+
"rstrip": false,
|
| 22 |
+
"single_word": false
|
| 23 |
+
}
|
| 24 |
+
}
|
pels_phi2_qlora/checkpoint-500/tokenizer.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
pels_phi2_qlora/checkpoint-500/tokenizer_config.json
ADDED
|
@@ -0,0 +1,325 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"add_prefix_space": false,
|
| 3 |
+
"added_tokens_decoder": {
|
| 4 |
+
"50256": {
|
| 5 |
+
"content": "<|endoftext|>",
|
| 6 |
+
"lstrip": false,
|
| 7 |
+
"normalized": false,
|
| 8 |
+
"rstrip": false,
|
| 9 |
+
"single_word": false,
|
| 10 |
+
"special": true
|
| 11 |
+
},
|
| 12 |
+
"50257": {
|
| 13 |
+
"content": " ",
|
| 14 |
+
"lstrip": false,
|
| 15 |
+
"normalized": true,
|
| 16 |
+
"rstrip": false,
|
| 17 |
+
"single_word": false,
|
| 18 |
+
"special": false
|
| 19 |
+
},
|
| 20 |
+
"50258": {
|
| 21 |
+
"content": " ",
|
| 22 |
+
"lstrip": false,
|
| 23 |
+
"normalized": true,
|
| 24 |
+
"rstrip": false,
|
| 25 |
+
"single_word": false,
|
| 26 |
+
"special": false
|
| 27 |
+
},
|
| 28 |
+
"50259": {
|
| 29 |
+
"content": " ",
|
| 30 |
+
"lstrip": false,
|
| 31 |
+
"normalized": true,
|
| 32 |
+
"rstrip": false,
|
| 33 |
+
"single_word": false,
|
| 34 |
+
"special": false
|
| 35 |
+
},
|
| 36 |
+
"50260": {
|
| 37 |
+
"content": " ",
|
| 38 |
+
"lstrip": false,
|
| 39 |
+
"normalized": true,
|
| 40 |
+
"rstrip": false,
|
| 41 |
+
"single_word": false,
|
| 42 |
+
"special": false
|
| 43 |
+
},
|
| 44 |
+
"50261": {
|
| 45 |
+
"content": " ",
|
| 46 |
+
"lstrip": false,
|
| 47 |
+
"normalized": true,
|
| 48 |
+
"rstrip": false,
|
| 49 |
+
"single_word": false,
|
| 50 |
+
"special": false
|
| 51 |
+
},
|
| 52 |
+
"50262": {
|
| 53 |
+
"content": " ",
|
| 54 |
+
"lstrip": false,
|
| 55 |
+
"normalized": true,
|
| 56 |
+
"rstrip": false,
|
| 57 |
+
"single_word": false,
|
| 58 |
+
"special": false
|
| 59 |
+
},
|
| 60 |
+
"50263": {
|
| 61 |
+
"content": " ",
|
| 62 |
+
"lstrip": false,
|
| 63 |
+
"normalized": true,
|
| 64 |
+
"rstrip": false,
|
| 65 |
+
"single_word": false,
|
| 66 |
+
"special": false
|
| 67 |
+
},
|
| 68 |
+
"50264": {
|
| 69 |
+
"content": " ",
|
| 70 |
+
"lstrip": false,
|
| 71 |
+
"normalized": true,
|
| 72 |
+
"rstrip": false,
|
| 73 |
+
"single_word": false,
|
| 74 |
+
"special": false
|
| 75 |
+
},
|
| 76 |
+
"50265": {
|
| 77 |
+
"content": " ",
|
| 78 |
+
"lstrip": false,
|
| 79 |
+
"normalized": true,
|
| 80 |
+
"rstrip": false,
|
| 81 |
+
"single_word": false,
|
| 82 |
+
"special": false
|
| 83 |
+
},
|
| 84 |
+
"50266": {
|
| 85 |
+
"content": " ",
|
| 86 |
+
"lstrip": false,
|
| 87 |
+
"normalized": true,
|
| 88 |
+
"rstrip": false,
|
| 89 |
+
"single_word": false,
|
| 90 |
+
"special": false
|
| 91 |
+
},
|
| 92 |
+
"50267": {
|
| 93 |
+
"content": " ",
|
| 94 |
+
"lstrip": false,
|
| 95 |
+
"normalized": true,
|
| 96 |
+
"rstrip": false,
|
| 97 |
+
"single_word": false,
|
| 98 |
+
"special": false
|
| 99 |
+
},
|
| 100 |
+
"50268": {
|
| 101 |
+
"content": " ",
|
| 102 |
+
"lstrip": false,
|
| 103 |
+
"normalized": true,
|
| 104 |
+
"rstrip": false,
|
| 105 |
+
"single_word": false,
|
| 106 |
+
"special": false
|
| 107 |
+
},
|
| 108 |
+
"50269": {
|
| 109 |
+
"content": " ",
|
| 110 |
+
"lstrip": false,
|
| 111 |
+
"normalized": true,
|
| 112 |
+
"rstrip": false,
|
| 113 |
+
"single_word": false,
|
| 114 |
+
"special": false
|
| 115 |
+
},
|
| 116 |
+
"50270": {
|
| 117 |
+
"content": " ",
|
| 118 |
+
"lstrip": false,
|
| 119 |
+
"normalized": true,
|
| 120 |
+
"rstrip": false,
|
| 121 |
+
"single_word": false,
|
| 122 |
+
"special": false
|
| 123 |
+
},
|
| 124 |
+
"50271": {
|
| 125 |
+
"content": " ",
|
| 126 |
+
"lstrip": false,
|
| 127 |
+
"normalized": true,
|
| 128 |
+
"rstrip": false,
|
| 129 |
+
"single_word": false,
|
| 130 |
+
"special": false
|
| 131 |
+
},
|
| 132 |
+
"50272": {
|
| 133 |
+
"content": " ",
|
| 134 |
+
"lstrip": false,
|
| 135 |
+
"normalized": true,
|
| 136 |
+
"rstrip": false,
|
| 137 |
+
"single_word": false,
|
| 138 |
+
"special": false
|
| 139 |
+
},
|
| 140 |
+
"50273": {
|
| 141 |
+
"content": " ",
|
| 142 |
+
"lstrip": false,
|
| 143 |
+
"normalized": true,
|
| 144 |
+
"rstrip": false,
|
| 145 |
+
"single_word": false,
|
| 146 |
+
"special": false
|
| 147 |
+
},
|
| 148 |
+
"50274": {
|
| 149 |
+
"content": " ",
|
| 150 |
+
"lstrip": false,
|
| 151 |
+
"normalized": true,
|
| 152 |
+
"rstrip": false,
|
| 153 |
+
"single_word": false,
|
| 154 |
+
"special": false
|
| 155 |
+
},
|
| 156 |
+
"50275": {
|
| 157 |
+
"content": " ",
|
| 158 |
+
"lstrip": false,
|
| 159 |
+
"normalized": true,
|
| 160 |
+
"rstrip": false,
|
| 161 |
+
"single_word": false,
|
| 162 |
+
"special": false
|
| 163 |
+
},
|
| 164 |
+
"50276": {
|
| 165 |
+
"content": " ",
|
| 166 |
+
"lstrip": false,
|
| 167 |
+
"normalized": true,
|
| 168 |
+
"rstrip": false,
|
| 169 |
+
"single_word": false,
|
| 170 |
+
"special": false
|
| 171 |
+
},
|
| 172 |
+
"50277": {
|
| 173 |
+
"content": " ",
|
| 174 |
+
"lstrip": false,
|
| 175 |
+
"normalized": true,
|
| 176 |
+
"rstrip": false,
|
| 177 |
+
"single_word": false,
|
| 178 |
+
"special": false
|
| 179 |
+
},
|
| 180 |
+
"50278": {
|
| 181 |
+
"content": " ",
|
| 182 |
+
"lstrip": false,
|
| 183 |
+
"normalized": true,
|
| 184 |
+
"rstrip": false,
|
| 185 |
+
"single_word": false,
|
| 186 |
+
"special": false
|
| 187 |
+
},
|
| 188 |
+
"50279": {
|
| 189 |
+
"content": " ",
|
| 190 |
+
"lstrip": false,
|
| 191 |
+
"normalized": true,
|
| 192 |
+
"rstrip": false,
|
| 193 |
+
"single_word": false,
|
| 194 |
+
"special": false
|
| 195 |
+
},
|
| 196 |
+
"50280": {
|
| 197 |
+
"content": " ",
|
| 198 |
+
"lstrip": false,
|
| 199 |
+
"normalized": true,
|
| 200 |
+
"rstrip": false,
|
| 201 |
+
"single_word": false,
|
| 202 |
+
"special": false
|
| 203 |
+
},
|
| 204 |
+
"50281": {
|
| 205 |
+
"content": " ",
|
| 206 |
+
"lstrip": false,
|
| 207 |
+
"normalized": true,
|
| 208 |
+
"rstrip": false,
|
| 209 |
+
"single_word": false,
|
| 210 |
+
"special": false
|
| 211 |
+
},
|
| 212 |
+
"50282": {
|
| 213 |
+
"content": " ",
|
| 214 |
+
"lstrip": false,
|
| 215 |
+
"normalized": true,
|
| 216 |
+
"rstrip": false,
|
| 217 |
+
"single_word": false,
|
| 218 |
+
"special": false
|
| 219 |
+
},
|
| 220 |
+
"50283": {
|
| 221 |
+
"content": " ",
|
| 222 |
+
"lstrip": false,
|
| 223 |
+
"normalized": true,
|
| 224 |
+
"rstrip": false,
|
| 225 |
+
"single_word": false,
|
| 226 |
+
"special": false
|
| 227 |
+
},
|
| 228 |
+
"50284": {
|
| 229 |
+
"content": " ",
|
| 230 |
+
"lstrip": false,
|
| 231 |
+
"normalized": true,
|
| 232 |
+
"rstrip": false,
|
| 233 |
+
"single_word": false,
|
| 234 |
+
"special": false
|
| 235 |
+
},
|
| 236 |
+
"50285": {
|
| 237 |
+
"content": " ",
|
| 238 |
+
"lstrip": false,
|
| 239 |
+
"normalized": true,
|
| 240 |
+
"rstrip": false,
|
| 241 |
+
"single_word": false,
|
| 242 |
+
"special": false
|
| 243 |
+
},
|
| 244 |
+
"50286": {
|
| 245 |
+
"content": " ",
|
| 246 |
+
"lstrip": false,
|
| 247 |
+
"normalized": true,
|
| 248 |
+
"rstrip": false,
|
| 249 |
+
"single_word": false,
|
| 250 |
+
"special": false
|
| 251 |
+
},
|
| 252 |
+
"50287": {
|
| 253 |
+
"content": "\t\t\t\t\t\t\t\t\t",
|
| 254 |
+
"lstrip": false,
|
| 255 |
+
"normalized": true,
|
| 256 |
+
"rstrip": false,
|
| 257 |
+
"single_word": false,
|
| 258 |
+
"special": false
|
| 259 |
+
},
|
| 260 |
+
"50288": {
|
| 261 |
+
"content": "\t\t\t\t\t\t\t\t",
|
| 262 |
+
"lstrip": false,
|
| 263 |
+
"normalized": true,
|
| 264 |
+
"rstrip": false,
|
| 265 |
+
"single_word": false,
|
| 266 |
+
"special": false
|
| 267 |
+
},
|
| 268 |
+
"50289": {
|
| 269 |
+
"content": "\t\t\t\t\t\t\t",
|
| 270 |
+
"lstrip": false,
|
| 271 |
+
"normalized": true,
|
| 272 |
+
"rstrip": false,
|
| 273 |
+
"single_word": false,
|
| 274 |
+
"special": false
|
| 275 |
+
},
|
| 276 |
+
"50290": {
|
| 277 |
+
"content": "\t\t\t\t\t\t",
|
| 278 |
+
"lstrip": false,
|
| 279 |
+
"normalized": true,
|
| 280 |
+
"rstrip": false,
|
| 281 |
+
"single_word": false,
|
| 282 |
+
"special": false
|
| 283 |
+
},
|
| 284 |
+
"50291": {
|
| 285 |
+
"content": "\t\t\t\t\t",
|
| 286 |
+
"lstrip": false,
|
| 287 |
+
"normalized": true,
|
| 288 |
+
"rstrip": false,
|
| 289 |
+
"single_word": false,
|
| 290 |
+
"special": false
|
| 291 |
+
},
|
| 292 |
+
"50292": {
|
| 293 |
+
"content": "\t\t\t\t",
|
| 294 |
+
"lstrip": false,
|
| 295 |
+
"normalized": true,
|
| 296 |
+
"rstrip": false,
|
| 297 |
+
"single_word": false,
|
| 298 |
+
"special": false
|
| 299 |
+
},
|
| 300 |
+
"50293": {
|
| 301 |
+
"content": "\t\t\t",
|
| 302 |
+
"lstrip": false,
|
| 303 |
+
"normalized": true,
|
| 304 |
+
"rstrip": false,
|
| 305 |
+
"single_word": false,
|
| 306 |
+
"special": false
|
| 307 |
+
},
|
| 308 |
+
"50294": {
|
| 309 |
+
"content": "\t\t",
|
| 310 |
+
"lstrip": false,
|
| 311 |
+
"normalized": true,
|
| 312 |
+
"rstrip": false,
|
| 313 |
+
"single_word": false,
|
| 314 |
+
"special": false
|
| 315 |
+
}
|
| 316 |
+
},
|
| 317 |
+
"bos_token": "<|endoftext|>",
|
| 318 |
+
"clean_up_tokenization_spaces": true,
|
| 319 |
+
"eos_token": "<|endoftext|>",
|
| 320 |
+
"model_max_length": 2048,
|
| 321 |
+
"pad_token": "<|endoftext|>",
|
| 322 |
+
"return_token_type_ids": false,
|
| 323 |
+
"tokenizer_class": "CodeGenTokenizer",
|
| 324 |
+
"unk_token": "<|endoftext|>"
|
| 325 |
+
}
|
pels_phi2_qlora/checkpoint-500/trainer_state.json
ADDED
|
@@ -0,0 +1,201 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"best_metric": 0.25585266947746277,
|
| 3 |
+
"best_model_checkpoint": "./pels_phi2_qlora\\checkpoint-500",
|
| 4 |
+
"epoch": 2.9850746268656714,
|
| 5 |
+
"eval_steps": 100,
|
| 6 |
+
"global_step": 500,
|
| 7 |
+
"is_hyper_param_search": false,
|
| 8 |
+
"is_local_process_zero": true,
|
| 9 |
+
"is_world_process_zero": true,
|
| 10 |
+
"log_history": [
|
| 11 |
+
{
|
| 12 |
+
"epoch": 0.14925373134328357,
|
| 13 |
+
"grad_norm": 1.4428153038024902,
|
| 14 |
+
"learning_rate": 0.00019230769230769233,
|
| 15 |
+
"loss": 2.6364,
|
| 16 |
+
"step": 25
|
| 17 |
+
},
|
| 18 |
+
{
|
| 19 |
+
"epoch": 0.29850746268656714,
|
| 20 |
+
"grad_norm": 0.8800177574157715,
|
| 21 |
+
"learning_rate": 0.00019874283308955057,
|
| 22 |
+
"loss": 1.0969,
|
| 23 |
+
"step": 50
|
| 24 |
+
},
|
| 25 |
+
{
|
| 26 |
+
"epoch": 0.44776119402985076,
|
| 27 |
+
"grad_norm": 0.8356912732124329,
|
| 28 |
+
"learning_rate": 0.0001947944062577507,
|
| 29 |
+
"loss": 0.6956,
|
| 30 |
+
"step": 75
|
| 31 |
+
},
|
| 32 |
+
{
|
| 33 |
+
"epoch": 0.5970149253731343,
|
| 34 |
+
"grad_norm": 0.660551905632019,
|
| 35 |
+
"learning_rate": 0.0001882602351338137,
|
| 36 |
+
"loss": 0.4576,
|
| 37 |
+
"step": 100
|
| 38 |
+
},
|
| 39 |
+
{
|
| 40 |
+
"epoch": 0.5970149253731343,
|
| 41 |
+
"eval_loss": 0.3502635955810547,
|
| 42 |
+
"eval_runtime": 116.3908,
|
| 43 |
+
"eval_samples_per_second": 1.435,
|
| 44 |
+
"eval_steps_per_second": 1.435,
|
| 45 |
+
"step": 100
|
| 46 |
+
},
|
| 47 |
+
{
|
| 48 |
+
"epoch": 0.746268656716418,
|
| 49 |
+
"grad_norm": 0.6952309608459473,
|
| 50 |
+
"learning_rate": 0.00017931855487268782,
|
| 51 |
+
"loss": 0.3221,
|
| 52 |
+
"step": 125
|
| 53 |
+
},
|
| 54 |
+
{
|
| 55 |
+
"epoch": 0.8955223880597015,
|
| 56 |
+
"grad_norm": 0.43301114439964294,
|
| 57 |
+
"learning_rate": 0.00016821327120267567,
|
| 58 |
+
"loss": 0.4255,
|
| 59 |
+
"step": 150
|
| 60 |
+
},
|
| 61 |
+
{
|
| 62 |
+
"epoch": 1.044776119402985,
|
| 63 |
+
"grad_norm": 0.3896200954914093,
|
| 64 |
+
"learning_rate": 0.00015524730731298134,
|
| 65 |
+
"loss": 0.2244,
|
| 66 |
+
"step": 175
|
| 67 |
+
},
|
| 68 |
+
{
|
| 69 |
+
"epoch": 1.1940298507462686,
|
| 70 |
+
"grad_norm": 0.33569812774658203,
|
| 71 |
+
"learning_rate": 0.00014077434089877037,
|
| 72 |
+
"loss": 0.3321,
|
| 73 |
+
"step": 200
|
| 74 |
+
},
|
| 75 |
+
{
|
| 76 |
+
"epoch": 1.1940298507462686,
|
| 77 |
+
"eval_loss": 0.27684858441352844,
|
| 78 |
+
"eval_runtime": 115.5816,
|
| 79 |
+
"eval_samples_per_second": 1.445,
|
| 80 |
+
"eval_steps_per_second": 1.445,
|
| 81 |
+
"step": 200
|
| 82 |
+
},
|
| 83 |
+
{
|
| 84 |
+
"epoch": 1.3432835820895521,
|
| 85 |
+
"grad_norm": 0.41767171025276184,
|
| 86 |
+
"learning_rate": 0.00012518915675561483,
|
| 87 |
+
"loss": 0.2908,
|
| 88 |
+
"step": 225
|
| 89 |
+
},
|
| 90 |
+
{
|
| 91 |
+
"epoch": 1.4925373134328357,
|
| 92 |
+
"grad_norm": 0.38486623764038086,
|
| 93 |
+
"learning_rate": 0.00010891687807939707,
|
| 94 |
+
"loss": 0.2308,
|
| 95 |
+
"step": 250
|
| 96 |
+
},
|
| 97 |
+
{
|
| 98 |
+
"epoch": 1.6417910447761193,
|
| 99 |
+
"grad_norm": 0.33858075737953186,
|
| 100 |
+
"learning_rate": 9.24013702137397e-05,
|
| 101 |
+
"loss": 0.3402,
|
| 102 |
+
"step": 275
|
| 103 |
+
},
|
| 104 |
+
{
|
| 105 |
+
"epoch": 1.7910447761194028,
|
| 106 |
+
"grad_norm": 0.37202343344688416,
|
| 107 |
+
"learning_rate": 7.6093133160502e-05,
|
| 108 |
+
"loss": 0.2394,
|
| 109 |
+
"step": 300
|
| 110 |
+
},
|
| 111 |
+
{
|
| 112 |
+
"epoch": 1.7910447761194028,
|
| 113 |
+
"eval_loss": 0.2645300626754761,
|
| 114 |
+
"eval_runtime": 117.2223,
|
| 115 |
+
"eval_samples_per_second": 1.425,
|
| 116 |
+
"eval_steps_per_second": 1.425,
|
| 117 |
+
"step": 300
|
| 118 |
+
},
|
| 119 |
+
{
|
| 120 |
+
"epoch": 1.9402985074626866,
|
| 121 |
+
"grad_norm": 0.2712254822254181,
|
| 122 |
+
"learning_rate": 6.0437013114095195e-05,
|
| 123 |
+
"loss": 0.323,
|
| 124 |
+
"step": 325
|
| 125 |
+
},
|
| 126 |
+
{
|
| 127 |
+
"epoch": 2.08955223880597,
|
| 128 |
+
"grad_norm": 0.3418550193309784,
|
| 129 |
+
"learning_rate": 4.58600682169262e-05,
|
| 130 |
+
"loss": 0.2754,
|
| 131 |
+
"step": 350
|
| 132 |
+
},
|
| 133 |
+
{
|
| 134 |
+
"epoch": 2.2388059701492535,
|
| 135 |
+
"grad_norm": 0.3196725845336914,
|
| 136 |
+
"learning_rate": 3.275991952653054e-05,
|
| 137 |
+
"loss": 0.2195,
|
| 138 |
+
"step": 375
|
| 139 |
+
},
|
| 140 |
+
{
|
| 141 |
+
"epoch": 2.388059701492537,
|
| 142 |
+
"grad_norm": 0.42392605543136597,
|
| 143 |
+
"learning_rate": 2.149390494964323e-05,
|
| 144 |
+
"loss": 0.3365,
|
| 145 |
+
"step": 400
|
| 146 |
+
},
|
| 147 |
+
{
|
| 148 |
+
"epoch": 2.388059701492537,
|
| 149 |
+
"eval_loss": 0.25696343183517456,
|
| 150 |
+
"eval_runtime": 116.1779,
|
| 151 |
+
"eval_samples_per_second": 1.437,
|
| 152 |
+
"eval_steps_per_second": 1.437,
|
| 153 |
+
"step": 400
|
| 154 |
+
},
|
| 155 |
+
{
|
| 156 |
+
"epoch": 2.5373134328358207,
|
| 157 |
+
"grad_norm": 0.482653945684433,
|
| 158 |
+
"learning_rate": 1.2369331995613653e-05,
|
| 159 |
+
"loss": 0.2168,
|
| 160 |
+
"step": 425
|
| 161 |
+
},
|
| 162 |
+
{
|
| 163 |
+
"epoch": 2.6865671641791042,
|
| 164 |
+
"grad_norm": 0.27890604734420776,
|
| 165 |
+
"learning_rate": 5.63509522864123e-06,
|
| 166 |
+
"loss": 0.3082,
|
| 167 |
+
"step": 450
|
| 168 |
+
},
|
| 169 |
+
{
|
| 170 |
+
"epoch": 2.835820895522388,
|
| 171 |
+
"grad_norm": 0.37825313210487366,
|
| 172 |
+
"learning_rate": 1.4748870728839347e-06,
|
| 173 |
+
"loss": 0.2584,
|
| 174 |
+
"step": 475
|
| 175 |
+
},
|
| 176 |
+
{
|
| 177 |
+
"epoch": 2.9850746268656714,
|
| 178 |
+
"grad_norm": 0.46783697605133057,
|
| 179 |
+
"learning_rate": 2.187161977540431e-09,
|
| 180 |
+
"loss": 0.1957,
|
| 181 |
+
"step": 500
|
| 182 |
+
},
|
| 183 |
+
{
|
| 184 |
+
"epoch": 2.9850746268656714,
|
| 185 |
+
"eval_loss": 0.25585266947746277,
|
| 186 |
+
"eval_runtime": 132.9521,
|
| 187 |
+
"eval_samples_per_second": 1.256,
|
| 188 |
+
"eval_steps_per_second": 1.256,
|
| 189 |
+
"step": 500
|
| 190 |
+
}
|
| 191 |
+
],
|
| 192 |
+
"logging_steps": 25,
|
| 193 |
+
"max_steps": 501,
|
| 194 |
+
"num_input_tokens_seen": 0,
|
| 195 |
+
"num_train_epochs": 3,
|
| 196 |
+
"save_steps": 100,
|
| 197 |
+
"total_flos": 3.238153764486144e+16,
|
| 198 |
+
"train_batch_size": 1,
|
| 199 |
+
"trial_name": null,
|
| 200 |
+
"trial_params": null
|
| 201 |
+
}
|
pels_phi2_qlora/checkpoint-500/training_args.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:ffd1259f93baa6dbe0764efae1c295a80e7889d53f7c7bfc7c725d52eabc931f
|
| 3 |
+
size 4984
|
pels_phi2_qlora/checkpoint-500/vocab.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
pels_phi2_qlora/merges.txt
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
pels_phi2_qlora/special_tokens_map.json
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"bos_token": {
|
| 3 |
+
"content": "<|endoftext|>",
|
| 4 |
+
"lstrip": false,
|
| 5 |
+
"normalized": false,
|
| 6 |
+
"rstrip": false,
|
| 7 |
+
"single_word": false
|
| 8 |
+
},
|
| 9 |
+
"eos_token": {
|
| 10 |
+
"content": "<|endoftext|>",
|
| 11 |
+
"lstrip": false,
|
| 12 |
+
"normalized": false,
|
| 13 |
+
"rstrip": false,
|
| 14 |
+
"single_word": false
|
| 15 |
+
},
|
| 16 |
+
"pad_token": "<|endoftext|>",
|
| 17 |
+
"unk_token": {
|
| 18 |
+
"content": "<|endoftext|>",
|
| 19 |
+
"lstrip": false,
|
| 20 |
+
"normalized": false,
|
| 21 |
+
"rstrip": false,
|
| 22 |
+
"single_word": false
|
| 23 |
+
}
|
| 24 |
+
}
|
pels_phi2_qlora/tokenizer.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
pels_phi2_qlora/tokenizer_config.json
ADDED
|
@@ -0,0 +1,325 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"add_prefix_space": false,
|
| 3 |
+
"added_tokens_decoder": {
|
| 4 |
+
"50256": {
|
| 5 |
+
"content": "<|endoftext|>",
|
| 6 |
+
"lstrip": false,
|
| 7 |
+
"normalized": false,
|
| 8 |
+
"rstrip": false,
|
| 9 |
+
"single_word": false,
|
| 10 |
+
"special": true
|
| 11 |
+
},
|
| 12 |
+
"50257": {
|
| 13 |
+
"content": " ",
|
| 14 |
+
"lstrip": false,
|
| 15 |
+
"normalized": true,
|
| 16 |
+
"rstrip": false,
|
| 17 |
+
"single_word": false,
|
| 18 |
+
"special": false
|
| 19 |
+
},
|
| 20 |
+
"50258": {
|
| 21 |
+
"content": " ",
|
| 22 |
+
"lstrip": false,
|
| 23 |
+
"normalized": true,
|
| 24 |
+
"rstrip": false,
|
| 25 |
+
"single_word": false,
|
| 26 |
+
"special": false
|
| 27 |
+
},
|
| 28 |
+
"50259": {
|
| 29 |
+
"content": " ",
|
| 30 |
+
"lstrip": false,
|
| 31 |
+
"normalized": true,
|
| 32 |
+
"rstrip": false,
|
| 33 |
+
"single_word": false,
|
| 34 |
+
"special": false
|
| 35 |
+
},
|
| 36 |
+
"50260": {
|
| 37 |
+
"content": " ",
|
| 38 |
+
"lstrip": false,
|
| 39 |
+
"normalized": true,
|
| 40 |
+
"rstrip": false,
|
| 41 |
+
"single_word": false,
|
| 42 |
+
"special": false
|
| 43 |
+
},
|
| 44 |
+
"50261": {
|
| 45 |
+
"content": " ",
|
| 46 |
+
"lstrip": false,
|
| 47 |
+
"normalized": true,
|
| 48 |
+
"rstrip": false,
|
| 49 |
+
"single_word": false,
|
| 50 |
+
"special": false
|
| 51 |
+
},
|
| 52 |
+
"50262": {
|
| 53 |
+
"content": " ",
|
| 54 |
+
"lstrip": false,
|
| 55 |
+
"normalized": true,
|
| 56 |
+
"rstrip": false,
|
| 57 |
+
"single_word": false,
|
| 58 |
+
"special": false
|
| 59 |
+
},
|
| 60 |
+
"50263": {
|
| 61 |
+
"content": " ",
|
| 62 |
+
"lstrip": false,
|
| 63 |
+
"normalized": true,
|
| 64 |
+
"rstrip": false,
|
| 65 |
+
"single_word": false,
|
| 66 |
+
"special": false
|
| 67 |
+
},
|
| 68 |
+
"50264": {
|
| 69 |
+
"content": " ",
|
| 70 |
+
"lstrip": false,
|
| 71 |
+
"normalized": true,
|
| 72 |
+
"rstrip": false,
|
| 73 |
+
"single_word": false,
|
| 74 |
+
"special": false
|
| 75 |
+
},
|
| 76 |
+
"50265": {
|
| 77 |
+
"content": " ",
|
| 78 |
+
"lstrip": false,
|
| 79 |
+
"normalized": true,
|
| 80 |
+
"rstrip": false,
|
| 81 |
+
"single_word": false,
|
| 82 |
+
"special": false
|
| 83 |
+
},
|
| 84 |
+
"50266": {
|
| 85 |
+
"content": " ",
|
| 86 |
+
"lstrip": false,
|
| 87 |
+
"normalized": true,
|
| 88 |
+
"rstrip": false,
|
| 89 |
+
"single_word": false,
|
| 90 |
+
"special": false
|
| 91 |
+
},
|
| 92 |
+
"50267": {
|
| 93 |
+
"content": " ",
|
| 94 |
+
"lstrip": false,
|
| 95 |
+
"normalized": true,
|
| 96 |
+
"rstrip": false,
|
| 97 |
+
"single_word": false,
|
| 98 |
+
"special": false
|
| 99 |
+
},
|
| 100 |
+
"50268": {
|
| 101 |
+
"content": " ",
|
| 102 |
+
"lstrip": false,
|
| 103 |
+
"normalized": true,
|
| 104 |
+
"rstrip": false,
|
| 105 |
+
"single_word": false,
|
| 106 |
+
"special": false
|
| 107 |
+
},
|
| 108 |
+
"50269": {
|
| 109 |
+
"content": " ",
|
| 110 |
+
"lstrip": false,
|
| 111 |
+
"normalized": true,
|
| 112 |
+
"rstrip": false,
|
| 113 |
+
"single_word": false,
|
| 114 |
+
"special": false
|
| 115 |
+
},
|
| 116 |
+
"50270": {
|
| 117 |
+
"content": " ",
|
| 118 |
+
"lstrip": false,
|
| 119 |
+
"normalized": true,
|
| 120 |
+
"rstrip": false,
|
| 121 |
+
"single_word": false,
|
| 122 |
+
"special": false
|
| 123 |
+
},
|
| 124 |
+
"50271": {
|
| 125 |
+
"content": " ",
|
| 126 |
+
"lstrip": false,
|
| 127 |
+
"normalized": true,
|
| 128 |
+
"rstrip": false,
|
| 129 |
+
"single_word": false,
|
| 130 |
+
"special": false
|
| 131 |
+
},
|
| 132 |
+
"50272": {
|
| 133 |
+
"content": " ",
|
| 134 |
+
"lstrip": false,
|
| 135 |
+
"normalized": true,
|
| 136 |
+
"rstrip": false,
|
| 137 |
+
"single_word": false,
|
| 138 |
+
"special": false
|
| 139 |
+
},
|
| 140 |
+
"50273": {
|
| 141 |
+
"content": " ",
|
| 142 |
+
"lstrip": false,
|
| 143 |
+
"normalized": true,
|
| 144 |
+
"rstrip": false,
|
| 145 |
+
"single_word": false,
|
| 146 |
+
"special": false
|
| 147 |
+
},
|
| 148 |
+
"50274": {
|
| 149 |
+
"content": " ",
|
| 150 |
+
"lstrip": false,
|
| 151 |
+
"normalized": true,
|
| 152 |
+
"rstrip": false,
|
| 153 |
+
"single_word": false,
|
| 154 |
+
"special": false
|
| 155 |
+
},
|
| 156 |
+
"50275": {
|
| 157 |
+
"content": " ",
|
| 158 |
+
"lstrip": false,
|
| 159 |
+
"normalized": true,
|
| 160 |
+
"rstrip": false,
|
| 161 |
+
"single_word": false,
|
| 162 |
+
"special": false
|
| 163 |
+
},
|
| 164 |
+
"50276": {
|
| 165 |
+
"content": " ",
|
| 166 |
+
"lstrip": false,
|
| 167 |
+
"normalized": true,
|
| 168 |
+
"rstrip": false,
|
| 169 |
+
"single_word": false,
|
| 170 |
+
"special": false
|
| 171 |
+
},
|
| 172 |
+
"50277": {
|
| 173 |
+
"content": " ",
|
| 174 |
+
"lstrip": false,
|
| 175 |
+
"normalized": true,
|
| 176 |
+
"rstrip": false,
|
| 177 |
+
"single_word": false,
|
| 178 |
+
"special": false
|
| 179 |
+
},
|
| 180 |
+
"50278": {
|
| 181 |
+
"content": " ",
|
| 182 |
+
"lstrip": false,
|
| 183 |
+
"normalized": true,
|
| 184 |
+
"rstrip": false,
|
| 185 |
+
"single_word": false,
|
| 186 |
+
"special": false
|
| 187 |
+
},
|
| 188 |
+
"50279": {
|
| 189 |
+
"content": " ",
|
| 190 |
+
"lstrip": false,
|
| 191 |
+
"normalized": true,
|
| 192 |
+
"rstrip": false,
|
| 193 |
+
"single_word": false,
|
| 194 |
+
"special": false
|
| 195 |
+
},
|
| 196 |
+
"50280": {
|
| 197 |
+
"content": " ",
|
| 198 |
+
"lstrip": false,
|
| 199 |
+
"normalized": true,
|
| 200 |
+
"rstrip": false,
|
| 201 |
+
"single_word": false,
|
| 202 |
+
"special": false
|
| 203 |
+
},
|
| 204 |
+
"50281": {
|
| 205 |
+
"content": " ",
|
| 206 |
+
"lstrip": false,
|
| 207 |
+
"normalized": true,
|
| 208 |
+
"rstrip": false,
|
| 209 |
+
"single_word": false,
|
| 210 |
+
"special": false
|
| 211 |
+
},
|
| 212 |
+
"50282": {
|
| 213 |
+
"content": " ",
|
| 214 |
+
"lstrip": false,
|
| 215 |
+
"normalized": true,
|
| 216 |
+
"rstrip": false,
|
| 217 |
+
"single_word": false,
|
| 218 |
+
"special": false
|
| 219 |
+
},
|
| 220 |
+
"50283": {
|
| 221 |
+
"content": " ",
|
| 222 |
+
"lstrip": false,
|
| 223 |
+
"normalized": true,
|
| 224 |
+
"rstrip": false,
|
| 225 |
+
"single_word": false,
|
| 226 |
+
"special": false
|
| 227 |
+
},
|
| 228 |
+
"50284": {
|
| 229 |
+
"content": " ",
|
| 230 |
+
"lstrip": false,
|
| 231 |
+
"normalized": true,
|
| 232 |
+
"rstrip": false,
|
| 233 |
+
"single_word": false,
|
| 234 |
+
"special": false
|
| 235 |
+
},
|
| 236 |
+
"50285": {
|
| 237 |
+
"content": " ",
|
| 238 |
+
"lstrip": false,
|
| 239 |
+
"normalized": true,
|
| 240 |
+
"rstrip": false,
|
| 241 |
+
"single_word": false,
|
| 242 |
+
"special": false
|
| 243 |
+
},
|
| 244 |
+
"50286": {
|
| 245 |
+
"content": " ",
|
| 246 |
+
"lstrip": false,
|
| 247 |
+
"normalized": true,
|
| 248 |
+
"rstrip": false,
|
| 249 |
+
"single_word": false,
|
| 250 |
+
"special": false
|
| 251 |
+
},
|
| 252 |
+
"50287": {
|
| 253 |
+
"content": "\t\t\t\t\t\t\t\t\t",
|
| 254 |
+
"lstrip": false,
|
| 255 |
+
"normalized": true,
|
| 256 |
+
"rstrip": false,
|
| 257 |
+
"single_word": false,
|
| 258 |
+
"special": false
|
| 259 |
+
},
|
| 260 |
+
"50288": {
|
| 261 |
+
"content": "\t\t\t\t\t\t\t\t",
|
| 262 |
+
"lstrip": false,
|
| 263 |
+
"normalized": true,
|
| 264 |
+
"rstrip": false,
|
| 265 |
+
"single_word": false,
|
| 266 |
+
"special": false
|
| 267 |
+
},
|
| 268 |
+
"50289": {
|
| 269 |
+
"content": "\t\t\t\t\t\t\t",
|
| 270 |
+
"lstrip": false,
|
| 271 |
+
"normalized": true,
|
| 272 |
+
"rstrip": false,
|
| 273 |
+
"single_word": false,
|
| 274 |
+
"special": false
|
| 275 |
+
},
|
| 276 |
+
"50290": {
|
| 277 |
+
"content": "\t\t\t\t\t\t",
|
| 278 |
+
"lstrip": false,
|
| 279 |
+
"normalized": true,
|
| 280 |
+
"rstrip": false,
|
| 281 |
+
"single_word": false,
|
| 282 |
+
"special": false
|
| 283 |
+
},
|
| 284 |
+
"50291": {
|
| 285 |
+
"content": "\t\t\t\t\t",
|
| 286 |
+
"lstrip": false,
|
| 287 |
+
"normalized": true,
|
| 288 |
+
"rstrip": false,
|
| 289 |
+
"single_word": false,
|
| 290 |
+
"special": false
|
| 291 |
+
},
|
| 292 |
+
"50292": {
|
| 293 |
+
"content": "\t\t\t\t",
|
| 294 |
+
"lstrip": false,
|
| 295 |
+
"normalized": true,
|
| 296 |
+
"rstrip": false,
|
| 297 |
+
"single_word": false,
|
| 298 |
+
"special": false
|
| 299 |
+
},
|
| 300 |
+
"50293": {
|
| 301 |
+
"content": "\t\t\t",
|
| 302 |
+
"lstrip": false,
|
| 303 |
+
"normalized": true,
|
| 304 |
+
"rstrip": false,
|
| 305 |
+
"single_word": false,
|
| 306 |
+
"special": false
|
| 307 |
+
},
|
| 308 |
+
"50294": {
|
| 309 |
+
"content": "\t\t",
|
| 310 |
+
"lstrip": false,
|
| 311 |
+
"normalized": true,
|
| 312 |
+
"rstrip": false,
|
| 313 |
+
"single_word": false,
|
| 314 |
+
"special": false
|
| 315 |
+
}
|
| 316 |
+
},
|
| 317 |
+
"bos_token": "<|endoftext|>",
|
| 318 |
+
"clean_up_tokenization_spaces": true,
|
| 319 |
+
"eos_token": "<|endoftext|>",
|
| 320 |
+
"model_max_length": 2048,
|
| 321 |
+
"pad_token": "<|endoftext|>",
|
| 322 |
+
"return_token_type_ids": false,
|
| 323 |
+
"tokenizer_class": "CodeGenTokenizer",
|
| 324 |
+
"unk_token": "<|endoftext|>"
|
| 325 |
+
}
|
pels_phi2_qlora/training_args.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:ffd1259f93baa6dbe0764efae1c295a80e7889d53f7c7bfc7c725d52eabc931f
|
| 3 |
+
size 4984
|
pels_phi2_qlora/vocab.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|