Spaces:
Sleeping
Sleeping
File size: 23,489 Bytes
f83b893 b92e0fa a5a4e31 3917449 f83b893 cdcdbc0 a5a4e31 f83b893 a5a4e31 87c662d f83b893 87c662d f83b893 87c662d f83b893 87c662d f83b893 87c662d a5a4e31 87c662d a5a4e31 87c662d a5a4e31 87c662d a5a4e31 87c662d a5a4e31 87c662d a5a4e31 f83b893 87c662d 2f164a7 87c662d 2f164a7 3917449 87c662d a5a4e31 87c662d a5a4e31 87c662d a5a4e31 87c662d a5a4e31 87c662d a5a4e31 87c662d a5a4e31 87c662d a5a4e31 87c662d a5a4e31 87c662d 7cf4036 87c662d 7cf4036 87c662d a5a4e31 87c662d 7cf4036 60bcad1 7cf4036 87c662d a5a4e31 f83b893 2f164a7 cdcdbc0 2f164a7 8075d5e 90d2f9c 2f164a7 90d2f9c cdcdbc0 90d2f9c 8075d5e 0998fe3 cdcdbc0 0998fe3 cdcdbc0 0998fe3 cdcdbc0 0998fe3 89ab26c 0998fe3 66b1fce 2f164a7 66b1fce 8075d5e 66b1fce 804d37d 66b1fce 0998fe3 a332434 f83b893 66b1fce a5a4e31 bcdf54b 66b1fce 90d2f9c 66b1fce 0998fe3 dd9f125 bcdf54b 66b1fce f83b893 bcdf54b 507d32c bcdf54b bc467fb bcdf54b bc467fb bcdf54b 146b423 bcdf54b accd87f | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 351 352 353 354 355 356 357 358 359 360 361 362 363 364 365 366 367 368 369 370 371 372 373 374 375 376 377 378 379 380 381 382 383 384 385 386 387 388 389 390 391 392 393 394 395 396 397 398 399 400 401 402 403 404 405 406 407 408 409 410 411 412 413 414 415 416 417 418 419 420 421 422 423 424 425 426 427 428 429 430 431 432 433 434 435 436 437 438 439 440 441 442 443 444 445 446 447 448 449 450 451 452 453 454 455 456 457 458 459 460 461 462 463 464 465 466 467 468 469 470 471 472 473 474 475 476 477 478 479 480 481 482 483 484 485 486 487 488 489 | import gradio as gr
from openai import OpenAI
import os
import base64
with open("HFImage.png", "rb") as f:
encoded = base64.b64encode(f.read()).decode()
IMAGE_HTML = f'<img src="data:image/png;base64,{encoded}" style="width:100%; border-radius:8px; margin-bottom:20px;" />'
with open("QuestionAnswerTeaser.png", "rb") as f:
encoded = base64.b64encode(f.read()).decode()
QA_IMAGE_HTML = f'<img src="data:image/png;base64,{encoded}" style="width:100%; border-radius:8px; margin-bottom:20px;" />'
with open("SummarisationTeaser.png", "rb") as f:
encoded = base64.b64encode(f.read()).decode()
SUMMARISATION_IMAGGE_HTML = f'<img src="data:image/png;base64,{encoded}" style="width:100%; border-radius:8px; margin-bottom:20px;" />'
with open("Text2ImageTeaser.png", "rb") as f:
encoded = base64.b64encode(f.read()).decode()
TEXT2IMAGE_IMAGE_HTML = f'<img src="data:image/png;base64,{encoded}" style="width:100%; border-radius:8px; margin-bottom:20px;" />'
with open("SyntheticImage.png", "rb") as f:
encoded = base64.b64encode(f.read()).decode()
SYNTHETIC_HTML = f'<img src="data:image/png;base64,{encoded}" style="width:100%; border-radius:8px; margin-bottom:20px;" />'
with open("RealImage.jpeg", "rb") as f:
encoded = base64.b64encode(f.read()).decode()
REAL_HTML = f'<img src="data:image/png;base64,{encoded}" style="width:100%; border-radius:8px; margin-bottom:20px;" />'
client = OpenAI(api_key=os.getenv("OPENAI_API_KEY"))
# ---- GPT explanation backend ----
def explain_text(selected_text):
if selected_text is None:
return ""
selected_text = selected_text.strip()
if not selected_text:
return "Please select or enter some text first."
try:
response = client.chat.completions.create(
model="gpt-4o",
messages=[
{"role": "system", "content": "You are an expert machine learning instructor. Explain concepts clearly and intuitively for learners with basic ML knowledge. Keep explanations concise and educational."},
{"role": "user", "content": f"Explain this text from a learning resource:\n\n\"\"\"\n{selected_text}\n\"\"\""}
],
temperature=0.7,
max_tokens=500
)
return response.choices[0].message.content
except Exception as e:
return f"Error: {str(e)}"
# ---- Your work content ----
YOUR_WORK_HTML = """
<div id="content" style="max-width: 800px; margin: auto; font-size: 16px; line-height: 1.6;">
<h1>Text Generation</h1>
<p>
Text generation is the task of producing natural language text given an input prompt.
It is commonly used for chatbots, creative writing, summarization, and code generation.
</p>
<p>
Most modern text generation models are based on the transformer architecture and are
trained using next-token prediction.
</p>
<p>
During inference, the model repeatedly samples the most likely next token until a
stopping condition is reached.
</p>
</div>
"""
# ---- Hugging Face reference content ----
TEXT_GENERATION = f"""
<div id="hf-content" style="max-width: 800px; margin: auto; font-size: 16px; line-height: 1.6;">
<h1>Text Generation (Hugging Face)</h1>
<p>
Generating text is the task of generating new text given another text.
These models can, for example, fill in incomplete text or paraphrase.
</p>
{IMAGE_HTML}
<h1>About Text Generation</h1>
<p>
This task covers guides on both <a href="https://huggingface.co/models?pipeline_tag=text-generation&sort=downloads">text-generation</a> and <a href="https://huggingface.co/models?other=text2text-generation&sort=downloads">text-to-text generation</a> models.
Popular large language models that are used for chats or following instructions are also covered in this task.
You can find the list of selected open-source large language models <a href="https://huggingface.co/spaces/open-llm-leaderboard/open_llm_leaderboard">here</a>, ranked by their performance scores.
</p>
<h2>Use Cases</h2>
<h3>Instruction Models</h3>
<p>
A model trained for text generation can be later adapted to follow instructions.
You can try some of the most powerful instruction-tuned open-access models like Mixtral 8x7B, Cohere Command R+, and Meta Llama3 70B at <a href="https://huggingface.co/chat">Hugging Chat</a>.
</p>
<h3>Code Generation</h3>
<p>
A Text Generation model, also known as a causal language model, can be trained on code from scratch to help the programmers in their repetitive coding tasks.
One of the most popular open-source models for code generation is StarCoder, which can generate code in 80+ languages. You can try it <a href="https://huggingface.co/spaces/bigcode/bigcode-playground">here</a>.
</p>
<h3>Stories Generation</h3>
<p>
A story generation model can receive an input like "Once upon a time" and proceed to create a story-like text based on those first words.
You can try <a href="https://huggingface.co/spaces/mosaicml/mpt-7b-storywriter">this application</a> which contains a model trained on story generation, by MosaicML.
If your generative model training data is different than your use case, you can train a causal language model from scratch.
Learn how to do it in the free transformers <a href="https://huggingface.co/course/chapter7/6?fw=pt">course</a>!
</p>
<h2>Task Variants</h2>
<h3>Completion Generation Models</h3>
<p>
A popular variant of Text Generation models predicts the next word given a bunch of words.
Word by word a longer text is formed that results in for example:
<ul>
<li>Given an incomplete sentence, complete it.
<li>Continue a story given the first sentences.
<li>Provided a code description, generate the code.
</ul>
The most popular models for this task are GPT-based models, Mistral or Llama series.
These models are trained on data that has no labels, so you just need plain text to train your own model.
You can train text generation models to generate a wide variety of documents, from code to stories.
</p>
<h3>Text-to-Text Generation Models</h3>
<p>
These models are trained to learn the mapping between a pair of texts (e.g. translation from one language to another).
The most popular variants of these models are NLLB, FLAN-T5, and BART.
Text-to-Text models are trained with multi-tasking capabilities, they can accomplish a wide range of tasks, including summarization, translation, and text classification.
</p>
<h3>Language Model Variants</h3>
When it comes to text generation, the underlying language model can come in several types:
<ul>
<li><b>Base models</b>: refers to plain language models like Mistral 7B and Meta Llama-3-70b. These models are good for fine-tuning and few-shot prompting.
<li><b>Instruction-trained models</b>: these models are trained in a multi-task manner to follow a broad range of instructions like "Write me a recipe for chocolate cake". Models like Qwen 2 7B, Yi 1.5 34B Chat, and Meta Llama 70B Instruct are examples of instruction-trained models. In general, instruction-trained models will produce better responses to instructions than base models.
<li><b>Human feedback models</b>: these models extend base and instruction-trained models by incorporating human feedback that rates the quality of the generated text according to criteria like helpfulness, honesty, and harmlessness. The human feedback is then combined with an optimization technique like reinforcement learning to align the original model to be closer with human preferences. The overall methodology is often called Reinforcement Learning from Human Feedback, or RLHF for short. Zephyr ORPO 141B A35B is an open-source model aligned through human feedback.
</ul>
<h2>Text Generation from Image and Text</h2>
<p>
There are language models that can input both text and image and output text, called vision language models.
IDEFICS 2 and MiniCPM Llama3 V are good examples.
They accept the same generation parameters as other language models.
However, since they also take images as input, you have to use them with the image-to-text pipeline.
You can find more information about this in the image-to-text task page.
</p>
<h2>Inference</h2>
<p>
You can use the 🤗 Transformers library <code>text-generation</code> pipeline to do inference
with text generation models. It takes an input text and generates a continuation of that text.
</p>
<pre style="background: #f5f5f5; padding: 15px; border-radius: 5px; overflow-x: auto;">
from transformers import pipeline
generator = pipeline('text-generation', model='gpt2')
generator("Hello, I'm a language model,", max_length=30, num_return_sequences=3)
</pre>
<p>
Text-to-Text generation models have a separate pipeline called text2text-generation.
This pipeline takes an input containing the sentence including the task and returns the output of the accomplished task.
<p>
""" + """<pre style="background: #f5f5f5; padding: 15px; border-radius: 5px; overflow-x: auto;">
from transformers import pipeline
text2text_generator = pipeline("text2text-generation")
text2text_generator("question: What is 42 ? context: 42 is the answer to life, the universe and everything")
[{'generated_text': 'the answer to life, the universe and everything'}]
text2text_generator("translate from English to French: I'm very happy")
[{'generated_text': 'Je suis très heureux'}]
</pre>
<p>
You can use huggingface.js to infer text classification models on Hugging Face Hub.
</p>
<pre style="background: #f5f5f5; padding: 15px; border-radius: 5px; overflow-x: auto;">
import { InferenceClient } from "@huggingface/inference";
const inference = new InferenceClient(HF_TOKEN);
await inference.conversational({
model: "distilbert-base-uncased-finetuned-sst-2-english",
inputs: "I love this movie!",
});
</pre>
<h2>Text Generation Inference</h2>
<p>
Text Generation Inference (TGI) is an open-source toolkit for serving LLMs tackling challenges such as response time.
TGI powers inference solutions like Inference Endpoints and Hugging Chat, as well as multiple community projects.
You can use it to deploy any supported open-source large language model of your choice.
</p>
<h2>ChatUI Spaces</h2>
<p>
Hugging Face Spaces includes templates to easily deploy your own instance of a specific application.
ChatUI is an open-source interface that enables serving conversational interface for large language models and can be deployed with few clicks at Spaces.
TGI powers these Spaces under the hood for faster inference.
Thanks to the template, you can deploy your own instance based on a large language model with only a few clicks and customize it. Learn more about it here and create your large language model instance here.
</div>
"""
QUESTION_ANSWER = f"""
<div id="hf-content" style="max-width: 800px; margin: auto; font-size: 16px; line-height: 1.6;">
<h1>Question Answering (Hugging Face)</h1>
<p>
Question Answering models can retrieve the answer to a question from a given text, which is useful for searching for an answer in a document.
Some question answering models can generate answers without context!
</p>
{QA_IMAGE_HTML}
<h1>About Question Answering</h1>
<h2>Use Cases</h2>
<h3>Frequently Asked Questions</h3>
<p>
You can use Question Answering (QA) models to automate the response to frequently asked questions by using a knowledge base (documents) as context.
Answers to customer questions can be drawn from those documents.
⚡⚡ If you’d like to save inference time, you can first use <a href="https://huggingface.co/tasks/sentence-similarity">passage ranking</a> models to see which document might contain the answer to the question and iterate over that document with the QA model instead.
</p>
<h2>Task Variants</h2>
<p>
There are different QA variants based on the inputs and outputs:
<ul>
<li><b>Extractive QA:</b> The model <b>extracts</b> the answer from a context.
The context here could be a provided text, a table or even HTML! This is usually solved with BERT-like models.</li>
<li><b>Open Generative QA:</b> The model <b>generates</b> free text directly based on the context.
You can learn more about the Text Generation task in its page.</li>
<li><b>Closed Generative QA:</b> In this case, no context is provided. The answer is completely generated by a model.</li>
</ul>
The schema above illustrates extractive, open book QA. The model takes a context and the question and extracts the answer from the given context.
You can also differentiate QA models depending on whether they are open-domain or closed-domain.
Open-domain models are not restricted to a specific domain, while closed-domain models are restricted to a specific domain (e.g. legal, medical documents).
</p>
<h2>Inference</h2>
<p>
You can infer with QA models with the 🤗 Transformers library using the question-answering pipeline.
If no model checkpoint is given, the pipeline will be initialized with distilbert-base-cased-distilled-squad.
This pipeline takes a question and a context from which the answer will be extracted and returned.
</p>
<pre style="background: #f5f5f5; padding: 15px; border-radius: 5px; overflow-x: auto;">
from transformers import pipeline
qa_model = pipeline("question-answering")
question = "Where do I live?"
context = "My name is Merve and I live in İstanbul."
qa_model(question = question, context = context)
## {{'answer': 'İstanbul', 'end': 39, 'score': 0.953, 'start': 31}}
</pre>
</div>
"""
SUMMARISATION = f"""
<div id="hf-content" style="max-width: 800px; margin: auto; font-size: 16px; line-height: 1.6;">
<h1>Summarisation (Hugging Face)</h1>
<p>
Summarization is the task of producing a shorter version of a document while preserving its important information.
Some models can extract text from the original input, while other models can generate entirely new text.
</p>
{SUMMARISATION_IMAGGE_HTML}
<h1>About Summarisation</h1>
<h2>Use Cases</h2>
<h3>Research Paper Summarization 🧐</h3>
<p>
Research papers can be summarized to allow researchers to spend less time selecting which articles to read.
There are several approaches you can take for a task like this:
<ol>
<li>Use an existing extractive summarization model on the Hub to do inference.</li>
<li>Pick an existing language model trained for academic papers. This model can then be trained in a process called fine-tuning so it can solve the summarization task.</li>
<li>Use a sequence-to-sequence model like T5 for abstractive text summarization.</li>
</ol>
</p>
<h3>Inference</h3>
<p>
You can use the 🤗 Transformers library summarization pipeline to infer with existing Summarization models.
If no model name is provided the pipeline will be initialized with <a href="https://huggingface.co/sshleifer/distilbart-cnn-12-6">sshleifer/distilbart-cnn-12-6</a>.
</p>
<pre style="background: #f5f5f5; padding: 15px; border-radius: 5px; overflow-x: auto;">
from transformers import pipeline
classifier = pipeline("summarization")
classifier("Paris is the capital and most populous city of France, with an estimated population of 2,175,601 residents as of 2018, in an area of more than 105 square kilometres (41 square miles). The City of Paris is the centre and seat of government of the region and province of Île-de-France, or Paris Region, which has an estimated population of 12,174,880, or about 18 percent of the population of France as of 2017.")
## [{{ "summary_text": " Paris is the capital and most populous city of France..." }}]
</pre>
<p>
You can use <a href="https://github.com/huggingface/huggingface.js">huggingface.js</a> to infer summarization models on Hugging Face Hub.
</p>
<pre style="background: #f5f5f5; padding: 15px; border-radius: 5px; overflow-x: auto;">
import {{ InferenceClient }} from "@huggingface/inference";
const inference = new InferenceClient(HF_TOKEN);
const inputs =
"Paris is the capital and most populous city of France, with an estimated population of 2,175,601 residents as of 2018, in an area of more than 105 square kilometres (41 square miles). The City of Paris is the centre and seat of government of the region and province of Île-de-France, or Paris Region, which has an estimated population of 12,174,880, or about 18 percent of the population of France as of 2017.";
await inference.summarization({{
model: "sshleifer/distilbart-cnn-12-6",
inputs,
}});
</pre>
</div>
"""
TEXT_2_IMAGGE = f"""
<div id="hf-content" style="max-width: 800px; margin: auto; font-size: 16px; line-height: 1.6;">
<h1>Text-to-Image (Hugging Face)</h1>
<p>
Text-to-image is the task of generating images from input text. These pipelines can also be used to modify and edit images based on text prompts.
</p>
{TEXT2IMAGE_IMAGE_HTML}
<h1>About Text-to-Image</h1>
<h2>Use Cases</h2>
<h3>Data Generation</h3>
<p>
Businesses can generate data for their use cases by inputting text and getting image outputs.
</p>
<h3>Immersive Conversational Chatbots</h3>
<p>
Chatbots can be made more immersive if they provide contextual images based on the input provided by the user.
</p>
<h3>Creative Ideas for Fashion Industry</h3>
<p>
Different patterns can be generated to obtain unique pieces of fashion.
Text-to-image models make creations easier for designers to conceptualize their design before actually implementing it.
</p>
<h3>Architecture Industry </h3>
<p>
Architects can utilise the models to construct an environment based out on the requirements of the floor plan.
This can also include the furniture that has to be placed in that environment.
</p>
<h2>Task Variants</h2>
<h3>Image Editing</h3>
<p>
Image editing with text-to-image models involves modifying an image following edit instructions provided in a text prompt.
<ul>
<li><b>Synthetic image editing:</b> Adjusting images that were initially created using an input prompt while preserving the overall meaning or context of the original image.</li>
</ul>
{SYNTHETIC_HTML}
<ul>
<li>Real image editing: Similar to synthetic image editing, except we're using real photos/images. This task is usually more complex.</li>
</ul>
{REAL_HTML}
</p>
<h3>Personalization</h3>
<p>
Personalization refers to techniques used to customize text-to-image models.
We introduce new subjects or concepts to the model, which the model can then generate when we refer to them with a text prompt.
For example, you can use these techniques to generate images of your dog in imaginary settings, after you have taught the model using a few reference images of the subject (or just one in some cases).
Teaching the model a new concept can be achieved through fine-tuning, or by using training-free techniques.
</p>
<h3>Inference</h3>
<p>
You can use diffusers pipelines to infer with text-to-image models.
</p>
<pre style="background: #f5f5f5; padding: 15px; border-radius: 5px; overflow-x: auto;">
from diffusers import StableDiffusionPipeline, EulerDiscreteScheduler
model_id = "stabilityai/stable-diffusion-2"
scheduler = EulerDiscreteScheduler.from_pretrained(model_id, subfolder="scheduler")
pipe = StableDiffusionPipeline.from_pretrained(model_id, scheduler=scheduler, torch_dtype=torch.float16)
pipe = pipe.to("cuda")
prompt = "a photo of an astronaut riding a horse on mars"
image = pipe(prompt).images[0]
</pre>
<p>
You can use <a href="https://github.com/huggingface/huggingface.js">huggingface.js</a> to infer text-to-image models on Hugging Face Hub.
</p>
<pre style="background: #f5f5f5; padding: 15px; border-radius: 5px; overflow-x: auto;">
import {{ InferenceClient }} from "@huggingface/inference";
const inference = new InferenceClient(HF_TOKEN);
await inference.textToImage({{
model: "stabilityai/stable-diffusion-2",
inputs: "award winning high resolution photo of a giant tortoise/((ladybird)) hybrid, [trending on artstation]",
parameters: {{
negative_prompt: "blurry",
}},
}});
</pre>
</div>
"""
# ---- Placeholder HTML pages ----
TEXT_GENERATION_HTML = TEXT_GENERATION
QUESTION_ANSWER_HTML = QUESTION_ANSWER
SUMMARISATION_HTML = SUMMARISATION
TEXT_2_IMAGGE_HTML = TEXT_2_IMAGGE
# ---- Page switching function ----
def switch_content(choice):
pages = {
"Text Generation": TEXT_GENERATION_HTML,
"Question Answering": QUESTION_ANSWER_HTML,
"Summarisation": SUMMARISATION_HTML,
"Text-to-Image": TEXT_2_IMAGGE_HTML,
}
return pages.get(choice, TEXT_GENERATION_HTML)
# =========================================================
# UI
# =========================================================
with gr.Blocks(head="""
<script>
document.addEventListener("mouseup", () => {
const selection = window.getSelection().toString().trim();
if (selection.length > 0) {
const textbox = document.querySelector(
'textarea[data-testid="textbox"]'
);
if (textbox) {
const nativeInputValueSetter =
Object.getOwnPropertyDescriptor(
window.HTMLTextAreaElement.prototype,
"value"
).set;
nativeInputValueSetter.call(textbox, selection);
textbox.dispatchEvent(
new Event("input", { bubbles: true })
);
}
}
});
</script>
""") as demo:
gr.Markdown("### 📘 Read through the resource before completing the rest of the survey")
# ---- Navigation bar ----
nav_bar = gr.Radio(
choices=[
"Text Generation",
"Question Answering",
"Summarisation",
"Text-to-Image",
],
value="Text Generation",
label="Topics",
interactive=True
)
# ---- Main content ----
content_display = gr.HTML(TEXT_GENERATION_HTML)
gr.HTML("""
<div style="
margin: 40px auto;
padding: 25px;
text-align: center;
font-size: 32px;
font-weight: bold;
border-radius: 12px;
">
Remember to return to the survey once you're done here!
</div>
""")
nav_bar.change(
fn=switch_content,
inputs=nav_bar,
outputs=content_display
)
demo.launch(allowed_paths=["."], share=True) |