{ "cells": [ { "cell_type": "code", "execution_count": null, "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "nYQNJK5hILPg", "outputId": "5442efec-88b0-4afc-a357-eaa3edb7b336" }, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ " Preparing metadata (setup.py) ... \u001b[?25l\u001b[?25hdone\n", "\u001b[2K \u001b[90m━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\u001b[0m \u001b[32m51.8/51.8 kB\u001b[0m \u001b[31m4.3 MB/s\u001b[0m eta \u001b[36m0:00:00\u001b[0m\n", "\u001b[2K \u001b[90m━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\u001b[0m \u001b[32m40.9/40.9 kB\u001b[0m \u001b[31m2.4 MB/s\u001b[0m eta \u001b[36m0:00:00\u001b[0m\n", "\u001b[2K \u001b[90m━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\u001b[0m \u001b[32m84.1/84.1 kB\u001b[0m \u001b[31m8.2 MB/s\u001b[0m eta \u001b[36m0:00:00\u001b[0m\n", "\u001b[2K \u001b[90m━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\u001b[0m \u001b[32m104.1/104.1 kB\u001b[0m \u001b[31m9.0 MB/s\u001b[0m eta \u001b[36m0:00:00\u001b[0m\n", "\u001b[2K \u001b[90m━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\u001b[0m \u001b[32m10.8/10.8 MB\u001b[0m \u001b[31m105.9 MB/s\u001b[0m eta \u001b[36m0:00:00\u001b[0m\n", "\u001b[2K \u001b[90m━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\u001b[0m \u001b[32m17.3/17.3 MB\u001b[0m \u001b[31m90.5 MB/s\u001b[0m eta \u001b[36m0:00:00\u001b[0m\n", "\u001b[2K \u001b[90m━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\u001b[0m \u001b[32m3.1/3.1 MB\u001b[0m \u001b[31m84.9 MB/s\u001b[0m eta \u001b[36m0:00:00\u001b[0m\n", "\u001b[2K \u001b[90m━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\u001b[0m \u001b[32m18.2/18.2 MB\u001b[0m \u001b[31m93.2 MB/s\u001b[0m eta \u001b[36m0:00:00\u001b[0m\n", "\u001b[2K \u001b[90m━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\u001b[0m \u001b[32m425.8/425.8 kB\u001b[0m \u001b[31m31.1 MB/s\u001b[0m eta \u001b[36m0:00:00\u001b[0m\n", "\u001b[2K \u001b[90m━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\u001b[0m \u001b[32m46.0/46.0 kB\u001b[0m \u001b[31m3.4 MB/s\u001b[0m eta \u001b[36m0:00:00\u001b[0m\n", "\u001b[2K \u001b[90m━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\u001b[0m \u001b[32m86.8/86.8 kB\u001b[0m \u001b[31m7.5 MB/s\u001b[0m eta \u001b[36m0:00:00\u001b[0m\n", "\u001b[?25h Building wheel for rouge-score (setup.py) ... \u001b[?25l\u001b[?25hdone\n" ] } ], "source": [ "!pip install -q transformers datasets peft accelerate evaluate rouge-score nltk sacrebleu torch optimum[onnxruntime]" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "7IZr99e2IRpS" }, "outputs": [], "source": [ "from datasets import load_dataset" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "colab": { "base_uri": "https://localhost:8080/", "height": 296, "referenced_widgets": [ "aecce17a6dd14284ac1b228c5bfc2c28", "e1b6c3a07c0c465dbfde5f65fd0a451d", "1f9d2261f13a4f90a69a238ebe762b17", "23e945fb213f4bb6abff49bb0eb081bd", "ff475050855c4524b7d598bcc52d3f69", "2366670474ad4fb4afb73c7a150cc4da", "3bcaa26b078b4fbc945a42bfd69ebbc0", "0bc2bb75acb64d90b81c96667b7b2349", "fe98d5ec1e334fa28816cc985414055e", "6baf0fe0e0de48028776aca87e20fcbb", "8423bb658f9943168901ca7b86cb3f7f", "4ee1d5f108aa4e5bb3e5b9eebb8b6417", "c0f0954cb281453591908cd24e62b6f2", "4c0747dbd717493ab11523d45a437920", "7260674b083f4457ba8d254ccf5fe1e8", "cb5282a788e74926a2146c60c3d7af3a", "25004871f40649039c582d2792d96336", "364bfd258ba941a5bb99dc8b006b1751", "0764b19c58234e8db7718362e76f70a7", "2e88d0e6d7c446378556d8aeec0f731d", "00d0c8e41765470c9cac83de7d2c6a88", "b46227d80ce9496cb2580953413bdc8f", "0437cfd15fbe4452a2dbdc3f6feabbce", "a1f362ef1a11420fb573d9f7c95921d4", "8b221f8b85b44494a45442137a1b3671", "6c3970a27484431b9463b4eb63e0a041", "ab657d4464eb4dd096a5e99e939eba2e", "68fe0cabb24e4cc3a5ae703aa3b35d04", "af4ee2833f5a44409e913d842bc28123", "d2c795b9ed094af8aeba9656f88f8b7b", "f47e92cd0a1148ccb6163ccbe2c33c57", "f06ae10cb1894fa1973da0688e2483a3", "4afcdccef3854391b017393d972d6e4b" ] }, "id": "scgQp7wEIWzZ", "outputId": "48958a9c-2837-40b8-d5e4-cbe743ddff63" }, "outputs": [ { "name": "stderr", "output_type": "stream", "text": [ "/usr/local/lib/python3.12/dist-packages/huggingface_hub/utils/_auth.py:94: UserWarning: \n", "The secret `HF_TOKEN` does not exist in your Colab secrets.\n", "To authenticate with the Hugging Face Hub, create a token in your settings tab (https://huggingface.co/settings/tokens), set it as secret in your Google Colab and restart your session.\n", "You will be able to reuse this secret in all of your notebooks.\n", "Please note that authentication is recommended but still optional to access public models or datasets.\n", " warnings.warn(\n" ] }, { "data": { "application/vnd.jupyter.widget-view+json": { "model_id": "aecce17a6dd14284ac1b228c5bfc2c28", "version_major": 2, "version_minor": 0 }, "text/plain": [ "README.md: 0%| | 0.00/118 [00:00, ?B/s]" ] }, "metadata": {}, "output_type": "display_data" }, { "data": { "application/vnd.jupyter.widget-view+json": { "model_id": "4ee1d5f108aa4e5bb3e5b9eebb8b6417", "version_major": 2, "version_minor": 0 }, "text/plain": [ "HealthCareMagic-100k.json: 0%| | 0.00/144M [00:00, ?B/s]" ] }, "metadata": {}, "output_type": "display_data" }, { "data": { "application/vnd.jupyter.widget-view+json": { "model_id": "0437cfd15fbe4452a2dbdc3f6feabbce", "version_major": 2, "version_minor": 0 }, "text/plain": [ "Generating train split: 0%| | 0/112165 [00:00, ? examples/s]" ] }, "metadata": {}, "output_type": "display_data" } ], "source": [ "dataset = load_dataset(\"BinKhoaLe1812/MedDialog-EN-100k\")" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "8CvSXtGoIbUU" }, "outputs": [], "source": [ "if 'validation' not in dataset:\n", " split_dataset = dataset['train'].train_test_split(test_size=0.1, seed=42)\n", " dataset['train'] = split_dataset['train']\n", " dataset['validation'] = split_dataset['test']" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "a06Xp5IUIfYD", "outputId": "b831d3b5-656c-4135-9f71-eef9bd7a097a" }, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "Training: 100,948\n", "Validation: 11,217\n" ] } ], "source": [ "print(f\"Training: {len(dataset['train']):,}\")\n", "print(f\"Validation: {len(dataset['validation']):,}\")" ] }, { "cell_type": "markdown", "metadata": { "id": "YonubsZ1Injq" }, "source": [ "### Load Tokenizer" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "0cNxBdZkI3bF", "outputId": "1b3229cd-c28d-4d33-adb0-6782278acf5b" }, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "Collecting sacremoses\n", " Downloading sacremoses-0.1.1-py3-none-any.whl.metadata (8.3 kB)\n", "Requirement already satisfied: regex in /usr/local/lib/python3.12/dist-packages (from sacremoses) (2024.11.6)\n", "Requirement already satisfied: click in /usr/local/lib/python3.12/dist-packages (from sacremoses) (8.3.0)\n", "Requirement already satisfied: joblib in /usr/local/lib/python3.12/dist-packages (from sacremoses) (1.5.2)\n", "Requirement already satisfied: tqdm in /usr/local/lib/python3.12/dist-packages (from sacremoses) (4.67.1)\n", "Downloading sacremoses-0.1.1-py3-none-any.whl (897 kB)\n", "\u001b[?25l \u001b[90m━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\u001b[0m \u001b[32m0.0/897.5 kB\u001b[0m \u001b[31m?\u001b[0m eta \u001b[36m-:--:--\u001b[0m\r\u001b[2K \u001b[90m━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\u001b[0m \u001b[32m897.5/897.5 kB\u001b[0m \u001b[31m27.2 MB/s\u001b[0m eta \u001b[36m0:00:00\u001b[0m\n", "\u001b[?25hInstalling collected packages: sacremoses\n", "Successfully installed sacremoses-0.1.1\n" ] } ], "source": [ "!pip install sacremoses" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "JXxUFBa_Iixp" }, "outputs": [], "source": [ "from transformers import AutoTokenizer" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "_ipeiBMtIrnR" }, "outputs": [], "source": [ "tokenizer = AutoTokenizer.from_pretrained(\"microsoft/biogpt\")" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "6A7008QuIv1V" }, "outputs": [], "source": [ "if tokenizer.pad_token is None:\n", " tokenizer.pad_token = tokenizer.eos_token" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "g-sl775iIwtr" }, "outputs": [], "source": [ "import torch\n", "from transformers import AutoModelForCausalLM" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "colab": { "base_uri": "https://localhost:8080/", "height": 81, "referenced_widgets": [ "e6d713586f65451ab2d073bdba63ced5", "66482d03eb7245ab80fe1737618d7649", "d3d64d6f723944249657e7bb56089c18", "1949ba303fe947d8bfc179d767f5b1b1", "1cda1547dde64775810fb2a94d9b7406", "1c7ea72eb3344fc79ec2f0a455e4cbc2", "511f9d02e0d04bd6b81b856bb15b062d", "f2e61cfbafd24894bff30b50831f0d72", "84cf01a9d6a14e0aa799d6d3f53c442b", "c5f864dddf1c41ca88ec710ec3c1c531", "6546bf03aec343dd9332e158c3285321", "db86783f3f6e409f9f825cd7819f9efe", "4ae77f4f79d545a6a61c40ae77845b96", "c5d49ae1537746eab37d0714d9eea786", "1c7368697110420e8c097842b3b4c0de", "b9a0d3013a4b46df835aa863635bbc4b", "aaa1416a06f04ac98d5182944a361010", "024460aece5e42f3a6a275f0ac97b815", "c557873434cc4c9abb7d7d6c403aafed", "01f972a9f19a46a18723b2457f98f596", "9df381957b90443690696129f8cb2eeb", "c1e87c30c63a4a308cd3c9c7d0c9f596" ] }, "id": "ccTfUCznI_xB", "outputId": "e595d75f-770b-4388-f3cb-370099297c60" }, "outputs": [ { "data": { "application/vnd.jupyter.widget-view+json": { "model_id": "e6d713586f65451ab2d073bdba63ced5", "version_major": 2, "version_minor": 0 }, "text/plain": [ "pytorch_model.bin: 0%| | 0.00/1.56G [00:00, ?B/s]" ] }, "metadata": {}, "output_type": "display_data" }, { "data": { "application/vnd.jupyter.widget-view+json": { "model_id": "db86783f3f6e409f9f825cd7819f9efe", "version_major": 2, "version_minor": 0 }, "text/plain": [ "model.safetensors: 0%| | 0.00/1.56G [00:00, ?B/s]" ] }, "metadata": {}, "output_type": "display_data" } ], "source": [ "model = AutoModelForCausalLM.from_pretrained(\n", " \"microsoft/biogpt\",\n", " torch_dtype=torch.float16,\n", " device_map=\"auto\",\n", ")" ] }, { "cell_type": "markdown", "metadata": { "id": "ROrY0A7sJKbU" }, "source": [ "### Setup LoRA" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "147GP7ibJL9H" }, "outputs": [], "source": [ "from peft import LoraConfig, get_peft_model, TaskType" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "sRStuMiCJPoK" }, "outputs": [], "source": [ "lora_config = LoraConfig(\n", " r=16, # Good rank for quality\n", " lora_alpha=32,\n", " lora_dropout=0.1,\n", " target_modules=[\"q_proj\", \"v_proj\", \"k_proj\", \"o_proj\"], # All 4 for quality\n", " bias=\"none\",\n", " task_type=TaskType.CAUSAL_LM,\n", ")" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "htwkBrcSJTGd" }, "outputs": [], "source": [ "model = get_peft_model(model, lora_config)\n", "model.enable_input_require_grads()" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "gibPnluPJVra" }, "outputs": [], "source": [ "trainable = sum(p.numel() for p in model.parameters() if p.requires_grad)\n", "total = sum(p.numel() for p in model.parameters())" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "GlQAxTosJYkp", "outputId": "8ada0097-c428-4156-b524-75b793e80ce7" }, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "Trainable: 2,359,296 (0.68%)\n" ] } ], "source": [ "print(f\"Trainable: {trainable:,} ({100*trainable/total:.2f}%)\")" ] }, { "cell_type": "markdown", "metadata": { "id": "S7OS_Pk_JmTP" }, "source": [ "Preprocess" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "DLws-KVuJkB8" }, "outputs": [], "source": [ "def preprocess_function(examples):\n", " texts = []\n", " for i in range(len(examples['instruction'])):\n", " instruction = examples['instruction'][i] or \"\"\n", " input_text = examples['input'][i] if 'input' in examples and examples['input'][i] else \"\"\n", " output_text = examples['output'][i] or \"\"\n", "\n", " prompt = f\"{instruction} {input_text}\".strip() if input_text else instruction.strip()\n", " text = f\"{prompt}\\n{output_text}\"\n", " texts.append(text)\n", "\n", " tokenized = tokenizer(\n", " texts,\n", " truncation=True,\n", " max_length=512, # Full length for quality\n", " padding=\"max_length\",\n", " return_tensors=\"pt\"\n", " )\n", " tokenized[\"labels\"] = tokenized[\"input_ids\"].clone()\n", " return tokenized" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "colab": { "base_uri": "https://localhost:8080/", "height": 77, "referenced_widgets": [ "9c661679d44d4f21a568d66727965333", "7b9970316f544e2ab6f906b1b53306d5", "6887d10706854ed6b22b2da2a015802b", "1ed4956dabd642cf9c64ae00b2035333", "ba69f31f140c4ebba9253ab30ac072b2", "ccf9db5c2f884a2f9f43c9661d9f62b9", "3c936c9da4df4680a98b2d0a095612b3", "fb519193a3cb4bf3a4bd3730ca4bb9cc", "c08c83c03062425a95283badc116c932", "1d8618f623b84e7c973fca29c5f5fd0f", "e7d86eaae07d493ea7d920478aad8878" ] }, "id": "b-d33MqLJqiB", "outputId": "d790a5f1-f3ab-468b-ddae-494196f0eb45" }, "outputs": [ { "data": { "application/vnd.jupyter.widget-view+json": { "model_id": "9c661679d44d4f21a568d66727965333", "version_major": 2, "version_minor": 0 }, "text/plain": [ "Train (num_proc=2): 0%| | 0/100948 [00:00, ? examples/s]" ] }, "metadata": {}, "output_type": "display_data" } ], "source": [ "tokenized_train = dataset['train'].map(\n", " preprocess_function,\n", " batched=True,\n", " remove_columns=dataset['train'].column_names,\n", " desc=\"Train\",\n", " num_proc=2\n", ")" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "colab": { "base_uri": "https://localhost:8080/", "height": 77, "referenced_widgets": [ "6783d566c96c469db6a61741c52ef116", "55a55dbb72fa47c99c632512a2eca3ca", "e665652135f4428f9695ea4d236d8337", "7f704b1af8564ca1b2fd37243d1a96c3", "b153f1968dfa42159c2cb69e153281ae", "76d84bcf96a4443798aaf4f1d2f21261", "debe4f6904a9496fb8d63860256ff1ff", "54dba99031ae4269a37fd8528645ae29", "b24252c137bb4e389eb6a412380aaf84", "3096ee80d5b444d2829617ce005b10d2", "750e9562c259413a9d40774ddb6237c1" ] }, "id": "BFCtrS4EJti1", "outputId": "9c62247d-46bc-40e6-eb0a-c2f0daf798e5" }, "outputs": [ { "data": { "application/vnd.jupyter.widget-view+json": { "model_id": "6783d566c96c469db6a61741c52ef116", "version_major": 2, "version_minor": 0 }, "text/plain": [ "Val (num_proc=2): 0%| | 0/11217 [00:00, ? examples/s]" ] }, "metadata": {}, "output_type": "display_data" } ], "source": [ "tokenized_eval = dataset['validation'].map(\n", " preprocess_function,\n", " batched=True,\n", " remove_columns=dataset['validation'].column_names,\n", " desc=\"Val\",\n", " num_proc=2\n", ")" ] }, { "cell_type": "markdown", "metadata": { "id": "Y8prciJPJ1KT" }, "source": [ "### Balanced Training Setup" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "5x5KT5caJxnG" }, "outputs": [], "source": [ "from transformers import TrainingArguments, Trainer, DataCollatorForLanguageModeling\n", "import numpy as np" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "cos0xDG8J5gW" }, "outputs": [], "source": [ "training_args = TrainingArguments(\n", " output_dir=\"./biogpt-lora-balanced\",\n", " per_device_train_batch_size=8, # Balanced\n", " per_device_eval_batch_size=8,\n", " gradient_accumulation_steps=2, # Effective batch = 16\n", " learning_rate=2e-4, # Standard LR\n", " num_train_epochs=1, # Full epochs for quality\n", " logging_steps=50,\n", " eval_strategy=\"steps\",\n", " eval_steps=500,\n", " save_strategy=\"steps\",\n", " save_steps=500,\n", " save_total_limit=2,\n", " load_best_model_at_end=True,\n", " metric_for_best_model=\"eval_loss\",\n", " greater_is_better=False,\n", " warmup_steps=100,\n", " fp16=True,\n", " gradient_checkpointing=True, # For memory\n", " dataloader_num_workers=2,\n", " dataloader_pin_memory=True,\n", " push_to_hub=False,\n", " report_to=\"none\",\n", ")" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "GSFwbb0aJ9tR" }, "outputs": [], "source": [ "data_collator = DataCollatorForLanguageModeling(tokenizer=tokenizer, mlm=False)" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "Aac4SxqMKA2t" }, "outputs": [], "source": [ "class PerplexityTrainer(Trainer):\n", " def evaluate(self, eval_dataset=None, ignore_keys=None, metric_key_prefix=\"eval\"):\n", " metrics = super().evaluate(eval_dataset, ignore_keys, metric_key_prefix)\n", " if f\"{metric_key_prefix}_loss\" in metrics:\n", " metrics[f\"{metric_key_prefix}_perplexity\"] = np.exp(metrics[f\"{metric_key_prefix}_loss\"])\n", " return metrics" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "lhMFqcvGKHaK", "outputId": "0716d9eb-9233-4ebf-aa54-f721c9ececef" }, "outputs": [ { "name": "stderr", "output_type": "stream", "text": [ "No label_names provided for model class `PeftModelForCausalLM`. Since `PeftModel` hides base models input arguments, if label_names is not given, label_names can't be set automatically within `Trainer`. Note that empty label_names list will be used instead.\n" ] } ], "source": [ "trainer = PerplexityTrainer(\n", " model=model,\n", " args=training_args,\n", " train_dataset=tokenized_train,\n", " eval_dataset=tokenized_eval,\n", " data_collator=data_collator,\n", ")" ] }, { "cell_type": "markdown", "metadata": { "id": "AKcCmUQsKKoJ" }, "source": [ "### Train" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "colab": { "background_save": true, "base_uri": "https://localhost:8080/", "height": 699 }, "id": "qil_8rSBKIUc", "outputId": "0dcd09c9-5ed3-4123-941f-2f6db7fc9ef5" }, "outputs": [ { "data": { "text/html": [ "\n", "
| Step | \n", "Training Loss | \n", "Validation Loss | \n", "
|---|---|---|
| 500 | \n", "3.359100 | \n", "3.246274 | \n", "
| 1000 | \n", "3.232400 | \n", "3.146441 | \n", "
| 1500 | \n", "3.186400 | \n", "3.092566 | \n", "
| 2000 | \n", "3.143200 | \n", "3.057678 | \n", "
| 2500 | \n", "3.138900 | \n", "3.031778 | \n", "
| 3000 | \n", "3.107300 | \n", "3.010683 | \n", "
| 3500 | \n", "3.104600 | \n", "2.998122 | \n", "
| 4000 | \n", "3.080200 | \n", "2.984522 | \n", "
| 4500 | \n", "3.082100 | \n", "2.975530 | \n", "
| 5000 | \n", "3.060300 | \n", "2.965659 | \n", "
| 5500 | \n", "3.053800 | \n", "2.962169 | \n", "
| 6000 | \n", "3.052300 | \n", "2.957626 | \n", "
"
],
"text/plain": [
"