{ "cells": [ { "cell_type": "markdown", "metadata": { "id": "LZ32HnEpxAyr" }, "source": [ "# Acknowledgments\n", "
\n", "\n", "\n", " \n", "
\n", "\n", "# Data Sources:\n", "* https://huggingface.co/angeluriot\n", "* https://huggingface.co/CATIE-AQ\n", "\n", "# Usefull plugins\n", "* jupyterlab-nvdashboard\n", "* jupyter-resource-usage\n", "\n", "# Transform into python script\n", "`jupyter nbconvert --to script Claire.ipynb`" ] }, { "cell_type": "markdown", "metadata": { "id": "rl_QlKk-zWv0" }, "source": [ "# Params\n", "- See: https://huggingface.co/learn/smol-course/unit1/3\n", "\n", "- Limited GPU Memory\n", " - per_device_train_batch_size = 2\n", " - gradient_accumulation_steps = 8\n", "- Balanced GPU Memory\n", " - per_device_train_batch_size = 4\n", " - gradient_accumulation_steps = 4\n", "- More GPU Memory\n", " - per_device_train_batch_size = 8\n", " - gradient_accumulation_steps = 2" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "editable": true, "id": "q7hxQ5anjxi_", "scrolled": true, "slideshow": { "slide_type": "" }, "tags": [] }, "outputs": [], "source": [ "import os\n", "\n", "if \"UNSLOTH_DOCKER\" in \"\".join(os.environ.keys()):\n", " hf_token = os.environ[\"HF_TOKEN\"]\n", " hf_username = os.environ[\"HF_USERNAME\"]\n", "elif \"COLAB_\" in \"\".join(os.environ.keys()):\n", " from google.colab import userdata\n", " hf_token = userdata.get('HuggingFaceToken')\n", " hf_username = userdata.get('HuggingFaceUsername')\n", "else:\n", " hf_token = os.environ[\"HF_TOKEN\"]\n", " hf_username = os.environ[\"HF_USERNAME\"]\n", " \n", "debug = False\n", "save_hf = True\n", "\n", "max_seq_length = 4096\n", "dataset_nb_processors = 8\n", "lora_rank = 16 # Choose any number > 0 ! Suggested 8, 16, 32, 64, 128\n", "training_learning_rate = 1e-4\n", "seed = 42\n", "training_per_device_train_batch_size = 2\n", "training_gradient_accumulation_steps = 8\n", "if debug: \n", " training_max_steps = 60\n", " training_num_train_epochs = 0\n", "else:\n", " training_max_steps = -1\n", " training_num_train_epochs = 1\n", "\n", "source_model = \"HuggingFaceTB/SmolLM3-3B\"\n", "source_model_name = \"SmolLM3-3B\"\n", "\n", "project_name = \"Claire-3B\"\n", "model_version = \"dev\" if debug else \"0.2.8\"\n", "model_name = project_name + \"-\" + model_version\n", "\n", "quantizations = [\"q8_0\", \"q4_0\", \"q4_k_m\", \"q5_k_m\"]" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "scrolled": true }, "outputs": [], "source": [ "import trackio\n", "if \"TRACKIO_SPACE_ID\" in \"\".join(os.environ.keys()):\n", " training_report_to = \"trackio\"\n", " training_space_id = os.environ[\"TRACKIO_SPACE_ID\"]\n", " # trackio.init(project=project_name.lower(), space_id=os.environ[\"TRACKIO_SPACE_ID\"])\n", "else:\n", " training_report_to = \"none\"\n", " training_space_id = None" ] }, { "cell_type": "markdown", "metadata": { "editable": true, "id": "84lHENVMzWv1", "slideshow": { "slide_type": "" }, "tags": [] }, "source": [ "# Installation" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "xJKunynhzWv1" }, "outputs": [], "source": [ "import os, re\n", "import torch\n", "\n", "if \"UNSLOTH_DOCKER\" in \"\".join(os.environ.keys()):\n", " # Docker Unsloth version\n", " pass\n", "elif \"COLAB_\" in \"\".join(os.environ.keys()):\n", " os.environ[\"PROTOCOL_BUFFERS_PYTHON_IMPLEMENTATION\"] = \"python\"\n", " # Google colab\n", " import torch; v = re.match(r\"[0-9\\.]{3,}\", str(torch.__version__)).group(0)\n", " xformers = \"xformers==\" + (\"0.0.32.post2\" if v == \"2.8.0\" else \"0.0.29.post3\")\n", " !pip install --upgrade --no-deps bitsandbytes accelerate {xformers} peft trl triton cut_cross_entropy unsloth_zoo\n", " !pip install sentencepiece protobuf \"datasets>=3.4.1,<4.0.0\" \"huggingface_hub>=0.34.0\" hf_transfer\n", " !pip install --no-deps --upgrade unsloth\n", " !pip install transformers==4.55.4\n", " !pip install --no-deps trl==0.22.2\n", "else:\n", " os.environ[\"PROTOCOL_BUFFERS_PYTHON_IMPLEMENTATION\"] = \"python\"\n", " !pip install --upgrade unsloth-zoo\n", " !pip install --upgrade unsloth\n", " !pip install transformers==4.55.4\n", " !pip install --no-deps trl==0.22.2\n", "\n", "from huggingface_hub import HfApi\n", "\n", "if save_hf:\n", " hf_api = HfApi(token=hf_token)\n", " hf_api.create_repo(repo_id = hf_username + \"/\" + model_name, repo_type = \"model\", private = True, exist_ok = True)\n", " hf_api.create_repo(repo_id = hf_username + \"/\" + model_name + \"-GGUF\", repo_type = \"model\", private = True, exist_ok = True)" ] }, { "cell_type": "markdown", "metadata": { "id": "sdsmJQ_8zrjP" }, "source": [ "# Tools\n" ] }, { "cell_type": "markdown", "metadata": { "editable": true, "id": "B6RdkRoFKYQT", "slideshow": { "slide_type": "" }, "tags": [] }, "source": [ "## Dataset formatter\n", "**[NOTE]** Remember to add the **EOS_TOKEN** to the tokenized output!! Otherwise you'll get infinite generations!\n", "\n", "ChatML renders multi turn conversations like below:\n", "\n", "```\n", "<|im_start|>system\n", "You are a helpful assistant.<|im_end|>\n", "<|im_start|>user\n", "What's the capital of France?<|im_end|>\n", "<|im_start|>assistant\n", "Paris.\n", "```" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "NhukxFpYy00v" }, "outputs": [], "source": [ "#@title Base Prompt\n", "wikipedia_prompt = \"\"\"Wikipedia Article\n", "### Title: {}\n", "\n", "### Article:\n", "{}\"\"\"\n", "\n", "ebook_prompt = \"\"\"Book\n", "### Title: {}\n", "\n", "### Author: {}\n", "\n", "### Content:\n", "{}\"\"\"" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "editable": true, "id": "_b0jOGAVyxFs", "slideshow": { "slide_type": "" }, "tags": [] }, "outputs": [], "source": [ "#@title SmallLm ChatML\n", "smollm_chatml = \"\"\"{# ───── defaults ───── #}\n", "{%- if enable_thinking is not defined -%}\n", "{%- set enable_thinking = true -%}\n", "{%- endif -%}\n", "\n", "{# ───── reasoning mode ───── #}\n", "{%- if enable_thinking -%}\n", " {%- set reasoning_mode = \"/think\" -%}\n", "{%- else -%}\n", " {%- set reasoning_mode = \"/no_think\" -%}\n", "{%- endif -%}\n", "\n", "{# ───── header (system message) ───── #}\n", "{{- \"<|im_start|>system\\n\" -}}\n", "\n", "{%- if messages[0].role == \"system\" -%}\n", " {%- set system_message = messages[0].content -%}\n", " {%- if \"/no_think\" in system_message -%}\n", " {%- set reasoning_mode = \"/no_think\" -%}\n", " {%- elif \"/think\" in system_message -%}\n", " {%- set reasoning_mode = \"/think\" -%}\n", " {%- endif -%}\n", " {%- set custom_instructions = system_message.replace(\"/no_think\", \"\").replace(\"/think\", \"\").rstrip() -%}\n", "{%- endif -%}\n", "\n", "{%- if \"/system_override\" in system_message -%}\n", " {{- custom_instructions.replace(\"/system_override\", \"\").rstrip() -}}\n", " {{- \"<|im_end|>\\n\" -}}\n", "{%- else -%}\n", " {{- \"## Metadata\\n\\n\" -}}\n", " {{- \"Knowledge Cutoff Date: June 2025\\n\" -}}\n", " {%- set today = strftime_now(\"%d %B %Y\") -%}\n", " {{- \"Today Date: \" ~ today ~ \"\\n\" -}}\n", " {{- \"Reasoning Mode: \" + reasoning_mode + \"\\n\\n\" -}}\n", " \n", " {{- \"## Custom Instructions\\n\\n\" -}}\n", " {%- if custom_instructions -%}\n", " {{- custom_instructions + \"\\n\\n\" -}}\n", " {%- elif reasoning_mode == \"/think\" -%}\n", " {{- \"You are a helpful AI assistant named Claire, trained by Hugging Face and fine tuned by Nathanaël SEMHOUN. Your role as an assistant involves thoroughly exploring questions through a systematic thinking process before providing the final precise and accurate solutions. This requires engaging in a comprehensive cycle of analysis, summarizing, exploration, reassessment, reflection, backtracking, and iteration to develop well-considered thinking process. Please structure your response into two main sections: Thought and Solution using the specified format: Thought section Solution section. In the Thought section, detail your reasoning process in steps. Each step should include detailed considerations such as analysing questions, summarizing relevant findings, brainstorming new ideas, verifying the accuracy of the current steps, refining any errors, and revisiting previous steps. In the Solution section, based on various attempts, explorations, and reflections from the Thought section, systematically present the final solution that you deem correct. The Solution section should be logical, accurate, and concise and detail necessary steps needed to reach the conclusion.\\n\\n\" -}}\n", " {%- else -%}\n", " {{- \"You are a helpful AI assistant named Claire, trained by Hugging Face and fine tuned by Nathanaël SEMHOUN.\\n\\n\" -}}\n", " {%- endif -%}\n", "\n", " {%- if xml_tools or python_tools or tools -%}\n", " {{- \"### Tools\\n\\n\" -}}\n", " {%- if xml_tools or tools -%}\n", " {%- if tools -%}\n", " {%- set xml_tools = tools -%}\n", " {%- endif -%}\n", " {%- set ns = namespace(xml_tool_string=\"You may call one or more functions to assist with the user query.\\nYou are provided with function signatures within XML tags:\\n\\n\\n\") -%}\n", " {%- for tool in xml_tools[:] -%} {# The slicing makes sure that xml_tools is a list #}\n", " {%- set ns.xml_tool_string = ns.xml_tool_string ~ (tool | string) ~ \"\\n\" -%}\n", " {%- endfor -%}\n", " {%- set xml_tool_string = ns.xml_tool_string + \"\\n\\nFor each function call, return a json object with function name and arguments within XML tags:\\n\\n\" + '{\"name\": , \"arguments\": }' + \"\\n\" -%}\n", " {{- xml_tool_string -}}\n", " {%- endif -%}\n", " {%- if python_tools -%}\n", " {%- set ns = namespace(python_tool_string=\"When you send a message containing Python code between '' and '' tags, it will be executed in a stateful Jupyter notebook environment, and you will then be given the output to continued reasoning in an agentic loop.\\n\\nYou can use the following tools in your python code like regular functions:\\n\\n\") -%}\n", " {%- for tool in python_tools[:] -%} {# The slicing makes sure that python_tools is a list #}\n", " {%- set ns.python_tool_string = ns.python_tool_string ~ (tool | string) ~ \"\\n\" -%}\n", " {%- endfor -%}\n", " {%- set python_tool_string = ns.python_tool_string + \"\\n\\nThe state persists between code executions: so variables that you define in one step are still available thereafter.\" -%}\n", " {{- python_tool_string -}}\n", " {%- endif -%}\n", " {{- \"\\n\\n\" -}}\n", " {{- \"<|im_end|>\\n\" -}}\n", " {%- endif -%}\n", "{%- endif -%}\n", "{# ───── main loop ───── #}\n", "{%- for message in messages -%}\n", " {%- set content = message.content if message.content is string else \"\" -%}\n", " {%- if message.role == \"user\" -%}\n", " {{ \"<|im_start|>\" + message.role + \"\\n\" + content + \"<|im_end|>\\n\" }}\n", " {%- elif message.role == \"assistant\" -%}\n", " {% generation %}\n", " {%- if reasoning_mode == \"/think\" -%}\n", " {{ \"<|im_start|>assistant\\n\" + content.lstrip(\"\\n\") + \"<|im_end|>\\n\" }}\n", " {%- else -%}\n", " {{ \"<|im_start|>assistant\\n\" + \"\\n\\n\\n\" + content.lstrip(\"\\n\") + \"<|im_end|>\\n\" }}\n", " {%- endif -%}\n", " {% endgeneration %}\n", " {%- elif message.role == \"tool\" -%}\n", " {{ \"<|im_start|>\" + \"user\\n\" + content + \"<|im_end|>\\n\" }}\n", " {%- endif -%}\n", "{%- endfor -%}\n", "{# ───── generation prompt ───── #}\n", "{%- if add_generation_prompt -%}\n", " {%- if reasoning_mode == \"/think\" -%}\n", " {{ \"<|im_start|>assistant\\n\" }}\n", " {%- else -%}\n", " {{ \"<|im_start|>assistant\\n\" + \"\\n\\n\\n\" }}\n", " {%- endif -%}\n", "{%- endif -%}\"\"\"\n", "smollm_eos='<|im_end|>'" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "editable": true, "id": "6xH9kHQ2zuzs", "slideshow": { "slide_type": "" }, "tags": [] }, "outputs": [], "source": [ "#@title Prompt Formatter\n", "import pprint, json\n", "\n", "def is_json(json_string):\n", " try:\n", " json.loads(json_string)\n", " return True\n", " except ValueError:\n", " return False\n", "\n", "def format_sft(examples, thinking):\n", " # convos or messages\n", " for key in (\"conversations\", \"conversation\", \"messages\"):\n", " convos = examples.get(key)\n", " if convos is not None:\n", " break\n", " else:\n", " raise ValueError(\"Unknown sft data format.\")\n", "\n", " # system and context\n", " for key in (\"system\", \"context\"):\n", " systems = examples.get(key)\n", " if systems is not None:\n", " # Special fix for context\n", " if (key == \"context\"):\n", " contexts = []\n", " for system in systems:\n", " if (len(system) > 5):\n", " contexts.append(\"Utilise le contexte entre pour répondre. Si tu ne sais pas, dis-le.\\n\\n\" + system +\"\\n\")\n", " else:\n", " contexts.append(\"\")\n", " systems = contexts\n", " break\n", " else:\n", " systems = [\"\"] * len(convos)\n", "\n", " if (examples.get('chat_template_kwargs') != None):\n", " kwargss = examples[\"chat_template_kwargs\"]\n", " else:\n", " kwargss = [{'enable_thinking' : thinking}] * len(convos) # Corrected list creation\n", "\n", " outputs = []\n", " for system, convo, kwargs in zip(systems, convos, kwargss):\n", " messages = [{\n", " \"role\": \"system\",\n", " \"content\": system\n", " }]\n", " for row in convo:\n", " role = row.get('from') or row.get('role')\n", " if role is None:\n", " raise ValueError('Unknow role data format.')\n", " normalized_role = str(role).lower()\n", " if 'humain' in normalized_role:\n", " role = 'user'\n", " elif 'gpt' in normalized_role:\n", " role = 'assistant'\n", " else:\n", " role = normalized_role\n", "\n", " content = row.get('content') or row.get('text')\n", " if content is None:\n", " raise ValueError('Unknow role data format.')\n", "\n", " messages.append({\n", " \"role\": role,\n", " \"content\": content\n", " })\n", " outputs.append(tokenizer.apply_chat_template(\n", " messages,\n", " tokenize=False,\n", " add_generation_prompt=False,\n", " **kwargs\n", " ))\n", " return { \"text\" : outputs, }\n", "\n", "def format_cpt_wikipedia(examples):\n", " titles = examples[\"title\"]\n", " texts = examples[\"text\"]\n", " outputs = []\n", " for title, text in zip(titles, texts):\n", " text = wikipedia_prompt.format(title, text) + EOS_TOKEN\n", " outputs.append(text)\n", " return { \"text\" : outputs, }\n", "\n", "def format_cpt_ebook(examples):\n", " titles = examples.get('title') or examples.get('titre')\n", " authors = examples.get('author') or examples.get('auteur')\n", " texts = examples.get('texte') or examples.get('content')\n", " outputs = []\n", " for title, author, text in zip(titles, authors, texts):\n", " text = ebook_prompt.format(title, author, text) + EOS_TOKEN\n", " outputs.append(text)\n", " return { \"text\" : outputs, }\n", "\n", "def format_text(examples):\n", " return { \"text\" : [example + EOS_TOKEN for example in examples[\"text\"]] }\n", "\n", "def format_dpo(examples, thinking):\n", " for key in (\"question\", \"prompt\"):\n", " prompts = examples.get(key)\n", " if prompts is not None:\n", " break\n", " else:\n", " raise ValueError(\"Unknown dpo data format.\")\n", "\n", " for key in (\"system\", \"context\"):\n", " systems = examples.get(key)\n", " if systems is not None:\n", " break\n", " else:\n", " systems = [\"\"] * len(prompts)\n", "\n", " chosens = examples.get('chosen')\n", " rejecteds = examples.get('rejected')\n", "\n", " outputs = {\n", " \"prompt\" : [],\n", " \"chosen\" : [],\n", " \"rejected\" : []\n", " }\n", " \n", " for system, prompt, chosen, rejected in zip(systems, prompts, chosens, rejecteds):\n", " messages = [{\n", " \"role\": \"system\",\n", " \"content\": system\n", " }, {\n", " \"role\": \"user\",\n", " \"content\": prompt\n", " }]\n", " kwargs = {'enable_thinking' : False}\n", " outputs[\"prompt\"].append(tokenizer.apply_chat_template(messages, tokenize=False, add_generation_prompt=True, **kwargs))\n", " outputs[\"chosen\"].append(chosen + EOS_TOKEN)\n", " outputs[\"rejected\"].append(rejected + EOS_TOKEN)\n", " return outputs\n", "\n", "# Must add EOS_TOKEN, otherwise your generation will go on forever!\n", "# batched=True is needed\n", "def formatting_prompts_func(\n", " examples,\n", " task: \"sft\", # \"sft\", \"cpt_wikipedia\", \"cpt_book\", \"text\", \"dpo\"\n", " thinking = False):\n", " formatting_functions = {\n", " 'sft': lambda exs: format_sft(exs, thinking),\n", " 'cpt_wikipedia': format_cpt_wikipedia,\n", " 'cpt_ebook': format_cpt_ebook,\n", " 'dpo': lambda exs: format_dpo(exs, thinking),\n", " 'text': format_text,\n", " }\n", " formatter = formatting_functions.get(task)\n", " if formatter:\n", " return formatter(examples)\n", " else:\n", " raise ValueError(f'Unknown task format: {task}')" ] }, { "cell_type": "markdown", "metadata": { "editable": true, "id": "4tBwTwc82YHY", "slideshow": { "slide_type": "" }, "tags": [] }, "source": [ "## Dataset helper" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "CgpsdsN02bAx" }, "outputs": [], "source": [ "import random\n", "from datasets import load_dataset\n", "from datasets import concatenate_datasets\n", "from datasets import Dataset\n", "\n", "def dumpDatasets(datasets):\n", " for key, value in datasets.items():\n", " print('=' * 25 + ' ' + key.upper() + ' ' + '=' * 25)\n", " row = value[random.randrange(0, len(value), 1)]\n", " for rkey, rvalue in row.items():\n", " print('#' * 10 + ' ' + rkey.upper())\n", " pprint.pp(rvalue[:4096])\n", " print(\"\\n\\n\")\n", "pass\n", "\n", "def mergeDatasets(datasets, test = True):\n", " resDataSet = Dataset.from_dict({})\n", " for key, dataset in datasets.items():\n", " dataset.shuffle()\n", " resDataSet = concatenate_datasets([resDataSet, dataset])\n", " if test:\n", " testSize = 0.005 if debug else 0.05\n", " resDataSet = resDataSet.train_test_split(test_size=testSize)\n", " return resDataSet\n", "pass\n", "\n", "def loadDataset(url, task, subset = None, thinking = False, split = 'train', partial = None):\n", " if (subset != None):\n", " dataset = load_dataset(url, subset, split = split, token = hf_token)\n", " else:\n", " dataset = load_dataset(url, split = split, token = hf_token)\n", " if debug:\n", " partial = 50 \n", " if partial != None:\n", " dataset = dataset.shuffle().select(range(partial))\n", " dataset = dataset.map(formatting_prompts_func, batched = True, fn_kwargs= {'task':task, 'thinking': thinking})\n", " if (task == 'dpo'):\n", " dataset = dataset.select_columns(['prompt', 'chosen', 'rejected'])\n", " else:\n", " dataset = dataset.select_columns(['text'])\n", " return dataset" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## Model Load /Save" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "from unsloth import FastLanguageModel\n", "from unsloth import is_bfloat16_supported\n", "from unsloth import FastModel\n", "from unsloth.chat_templates import get_chat_template\n", "\n", "def loadModel(modelToLoad, embeded = False):\n", " global EOS_TOKEN\n", " \n", " model, tokenizer = FastLanguageModel.from_pretrained(\n", " model_name = source_model,\n", " max_seq_length = max_seq_length,\n", " dtype = None, # None for auto detection. Float16 for Tesla T4, V100, Bfloat16 for Ampere+\n", " load_in_4bit = debug,\n", " load_in_8bit = False,\n", " token = hf_token,\n", " )\n", " if embeded:\n", " model = FastLanguageModel.get_peft_model(\n", " model,\n", " r = lora_rank,\n", " target_modules = [\n", " \"q_proj\", \"k_proj\", \"v_proj\", \"o_proj\",\n", " \"gate_proj\", \"up_proj\", \"down_proj\"\n", " \"embed_tokens\", \"lm_head\", # Add for continual pretraining, it seem to be replaced by module_to_save\n", " ],\n", " modules_to_save=[\"embed_tokens\", \"lm_head\"],\n", " lora_alpha = lora_rank * 2, # *2 speeds up training\n", " lora_dropout = 0, # Supports any, but = 0 is optimized\n", " bias = \"none\", # Supports any, but = \"none\" is optimized\n", " use_gradient_checkpointing = \"unsloth\", # True or \"unsloth\" for very long context\n", " random_state = 3407,\n", " use_rslora = False, # We support rank stabilized LoRA\n", " loftq_config = None, # And LoftQ\n", " )\n", " else:\n", " model = FastLanguageModel.get_peft_model(\n", " model,\n", " r = lora_rank,\n", " target_modules = [\n", " \"q_proj\", \"k_proj\", \"v_proj\", \"o_proj\",\n", " \"gate_proj\", \"up_proj\", \"down_proj\"\n", " ],\n", " lora_alpha = lora_rank * 2, # *2 speeds up training\n", " lora_dropout = 0, # Supports any, but = 0 is optimized\n", " bias = \"none\", # Supports any, but = \"none\" is optimized\n", " use_gradient_checkpointing = \"unsloth\", # True or \"unsloth\" for very long context\n", " random_state = 3407,\n", " use_rslora = False, # We support rank stabilized LoRA\n", " loftq_config = None, # And LoftQ\n", " )\n", "\n", " tokenizer = get_chat_template(\n", " tokenizer,\n", " chat_template = (smollm_chatml, smollm_eos,), # You must provide a template and EOS token\n", " map_eos_token = True, # Maps <|im_end|> to instead\n", " )\n", " EOS_TOKEN = tokenizer.eos_token\n", " \n", " return model, tokenizer\n", "\n", "def saveModel(model_name, quantizations, hf_repo = None):\n", " model.save_pretrained_merged(model_name, tokenizer)\n", " model.save_pretrained_gguf(model_name, tokenizer, quantization_method = quantizations)\n", " \n", " if save_hf:\n", " hf_dest = hf_username + \"/\" + model_name\n", " if hf_repo != None:\n", " hf_dest = hf_username + \"/\" + hf_repo\n", " model.push_to_hub(hf_dest, token = hf_token)\n", " tokenizer.push_to_hub(hf_dest, token = hf_token)\n", " model.push_to_hub_merged(hf_dest, token = hf_token) \n", " for quant in quantizations:\n", " quant = quant.upper()\n", " hf_api.upload_file(\n", " path_or_fileobj=source_model_name + \".\" + quant + \".gguf\",\n", " path_in_repo=model_name + \"-\" + quant + \".gguf\",\n", " repo_id=hf_dest + \"-GGUF\"\n", " )" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## Testing (inference)" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "def testModel():\n", " FastLanguageModel.for_inference(model) # Enable native 2x faster inference\n", " from transformers import TextStreamer\n", " text_streamer = TextStreamer(tokenizer)\n", " \n", " tests = [\n", " [\n", " {\"role\": \"system\", \"content\": \"/no_think\"},\n", " {\"role\": \"user\", \"content\": \"Bonjour, Parle moi de toi.\"},\n", " ],\n", " [\n", " {\"role\": \"system\", \"content\": \"/no_think\"},\n", " {\"role\": \"user\", \"content\": \"Continue la séquence de fibonnaci (juste 10 nombres): 1, 1, 2, 3, 5, 8,\"},\n", " ],\n", " [\n", " {\"role\": \"system\", \"content\": \"/no_think\"},\n", " {\"role\": \"user\", \"content\": \"Quel est la fameuse grande tour à Paris?\"},\n", " ],\n", " [\n", " {\"role\": \"system\", \"content\": \"/think\"},\n", " {\"role\": \"user\", \"content\": \"Quel est la fameuse grande tour à Paris?\"},\n", " ],\n", " [\n", " {\"role\": \"system\", \"content\": \"/no_think\"},\n", " {\"role\": \"user\", \"content\": \"Explique moi le théorème de Pythagore\"},\n", " ],\n", " [\n", " {\"role\": \"system\", \"content\": \"Tu es un professeur de mathématique au collège.\\n/think\"},\n", " {\"role\": \"user\", \"content\": \"Explique moi le théorème de Pythagore\"},\n", " ],\n", " [\n", " {\"role\": \"system\", \"content\": \"/think\"},\n", " {\"role\": \"user\", \"content\": \"Résume moi le premier livre de la sage Dune écrite par Frank Herbert\"},\n", " ],\n", " ]\n", " \n", " for messages in tests:\n", " print(80 * \"=\")\n", " inputs = tokenizer.apply_chat_template(\n", " messages,\n", " tokenize = True,\n", " add_generation_prompt = True, # Must add for generation\n", " return_tensors = \"pt\",\n", " ).to(\"cuda\")\n", " model.generate(input_ids = inputs, streamer = text_streamer, max_new_tokens = 2048, use_cache = False)\n", " print(\"\\n\\n\")" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "# Base Model" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "model, tokenizer = loadModel(source_model, True)" ] }, { "cell_type": "markdown", "metadata": { "id": "VHHZqnHe5NFG" }, "source": [ "# Continued pre-training (CPT)" ] }, { "cell_type": "markdown", "metadata": { "id": "hyBhe7yBBzZs" }, "source": [ "## Datasets(s)\n", "\n" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "NkLEJDX0P4hh" }, "outputs": [], "source": [ "#@title Init datasets\n", "datasets = {}" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "Hjt-pJC1Tj_q", "scrolled": true }, "outputs": [], "source": [ "#@title EBook data\n", "datasets['ebooks'] = loadDataset('nsemhoun/ebooks', 'cpt_ebook')" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "4yZXzs63PBKY", "scrolled": true }, "outputs": [], "source": [ "#@title Wikipedia data\n", "if False: datasets['wikipedia'] = loadDataset('omarkamali/wikipedia-monthly', 'cpt_wikipedia', subset='latest.fr')" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "#@title Gutenberg data\n", "datasets['gutenberg'] = loadDataset('CATIE-AQ/french_books', 'cpt_ebook')" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "IFZt5e4e2vGR", "scrolled": true }, "outputs": [], "source": [ "#@title Dump datasets\n", "dumpDatasets(datasets)" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "r_Ze9iDzPyms" }, "outputs": [], "source": [ "#@title Merge datasets\n", "dataset = mergeDatasets(datasets, test = False)" ] }, { "cell_type": "markdown", "metadata": { "editable": true, "id": "wWLc335hC2Yy", "slideshow": { "slide_type": "" }, "tags": [] }, "source": [ "## Training" ] }, { "cell_type": "markdown", "metadata": { "id": "5EE6r_wozEan" }, "source": [ "### Setup Trainer" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "ntZ2UAJwC9-F", "scrolled": true }, "outputs": [], "source": [ "from unsloth import UnslothTrainer, UnslothTrainingArguments\n", "\n", "trainer = UnslothTrainer(\n", " model = model,\n", " tokenizer = tokenizer,\n", " train_dataset = dataset,\n", " dataset_text_field = \"text\",\n", " max_seq_length = max_seq_length,\n", "\n", " chat_template = None,\n", " \n", " args = UnslothTrainingArguments( \n", " per_device_train_batch_size = training_per_device_train_batch_size,\n", " gradient_accumulation_steps = training_gradient_accumulation_steps,\n", " \n", " dataset_num_proc = dataset_nb_processors,\n", " \n", " max_steps = training_max_steps,\n", " num_train_epochs = training_num_train_epochs,\n", " warmup_ratio = 0.1,\n", " \n", " learning_rate = training_learning_rate,\n", " embedding_learning_rate = (training_learning_rate / 10),\n", "\n", " logging_steps = 1,\n", " optim = \"adamw_8bit\",\n", " weight_decay = 0.01,\n", " lr_scheduler_type = \"linear\",\n", " seed = seed,\n", " \n", " output_dir = \"outputs\",\n", " report_to = training_report_to,\n", " trackio_space_id = training_space_id,\n", " run_name = project_name.lower() + '-' + model_version + \"-cpt\",\n", " ),\n", ")" ] }, { "cell_type": "markdown", "metadata": { "editable": true, "id": "Z4sot7NEzNYs", "slideshow": { "slide_type": "" }, "tags": [] }, "source": [ "### Training Execution\n", "Execute the training process with the configured trainer and monitor the training progress." ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "mfQ_U1kgDq43" }, "outputs": [], "source": [ "gpu_stats = torch.cuda.get_device_properties(0)\n", "start_gpu_memory = round(torch.cuda.max_memory_reserved() / 1024 / 1024 / 1024, 3)\n", "max_memory = round(gpu_stats.total_memory / 1024 / 1024 / 1024, 3)\n", "print(f\"GPU = {gpu_stats.name}. Max memory = {max_memory} GB.\")\n", "print(f\"{start_gpu_memory} GB of memory reserved.\")" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "DvaF_VJiDsWK", "scrolled": true }, "outputs": [], "source": [ "trainer_stats = trainer.train()" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "txSvDDqeDvIS" }, "outputs": [], "source": [ "used_memory = round(torch.cuda.max_memory_reserved() / 1024 / 1024 / 1024, 3)\n", "used_memory_for_lora = round(used_memory - start_gpu_memory, 3)\n", "used_percentage = round(used_memory /max_memory*100, 3)\n", "lora_percentage = round(used_memory_for_lora/max_memory*100, 3)\n", "print(f\"{trainer_stats.metrics['train_runtime']} seconds used for training.\")\n", "print(f\"{round(trainer_stats.metrics['train_runtime']/60, 2)} minutes used for training.\")\n", "print(f\"Peak reserved memory = {used_memory} GB.\")\n", "print(f\"Peak reserved memory for training = {used_memory_for_lora} GB.\")\n", "print(f\"Peak reserved memory % of max memory = {used_percentage} %.\")\n", "print(f\"Peak reserved memory for training % of max memory = {lora_percentage} %.\")" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## Save and test" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "scrolled": true }, "outputs": [], "source": [ "saveModel(model_name + \"-cpt\", ['q4_k_m'], hf_repo = project_name)\n", "testModel()" ] }, { "cell_type": "markdown", "metadata": { "id": "WlYPClPP32Ue" }, "source": [ "# Supervised Fine-Tuning (SFT)" ] }, { "cell_type": "markdown", "metadata": { "editable": true, "id": "898-nEBcgrlC", "slideshow": { "slide_type": "" }, "tags": [] }, "source": [ "## Datasets(s)" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "zi3DWL96gwtq" }, "outputs": [], "source": [ "#@title Init datasets\n", "datasets = {}" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "editable": true, "id": "4ULTH9QphZlK", "slideshow": { "slide_type": "" }, "tags": [] }, "outputs": [], "source": [ "#@title SmolTalk Think\n", "datasets['smaltalk-think'] = loadDataset('CATIE-AQ/smoltalk2_aya_think_dataset_french_split', 'sft', subset='french_raisonning', thinking=True)" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "iULedxTFh1G4" }, "outputs": [], "source": [ "#@title SmolTalk ToolCalling\n", "datasets['smaltalk-toolcalling'] = loadDataset('CATIE-AQ/smoltalk2_smolagents_toolcalling_french', 'sft')" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "t490pJK7iJDG" }, "outputs": [], "source": [ "#@title Facebook Community\n", "datasets['facebook'] = loadDataset('CATIE-AQ/facebook-community-alignment-dataset_french_conversation', 'sft', partial = 8192)" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "#@title Everyday conversation\n", "datasets['everyday'] = loadDataset('CATIE-AQ/everyday-conversations-llama3.1-2k-in-french', 'sft')" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "editable": true, "id": "TmOhhrQS0QYZ", "scrolled": true, "slideshow": { "slide_type": "" }, "tags": [] }, "outputs": [], "source": [ "#@title Dump datasets\n", "dumpDatasets(datasets)" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "#@title Merge datasets\n", "dataset = mergeDatasets(datasets)" ] }, { "cell_type": "markdown", "metadata": { "editable": true, "id": "idAEIeSQ3xdS", "slideshow": { "slide_type": "" }, "tags": [] }, "source": [ "## Training\n" ] }, { "cell_type": "markdown", "metadata": { "id": "IAsG0QDAWdZF" }, "source": [ "### Setup Trainer" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "editable": true, "id": "95_Nn-89DhsL", "slideshow": { "slide_type": "" }, "tags": [] }, "outputs": [], "source": [ "from trl import SFTTrainer, SFTConfig\n", "from unsloth import is_bfloat16_supported\n", "\n", "trainer = SFTTrainer(\n", " model = model,\n", " tokenizer = tokenizer,\n", " train_dataset = dataset['train'],\n", " eval_dataset = dataset['test'],\n", " dataset_text_field = \"text\",\n", " max_seq_length = max_seq_length,\n", " \n", " args = SFTConfig(\n", " per_device_train_batch_size = training_per_device_train_batch_size,\n", " gradient_accumulation_steps = training_gradient_accumulation_steps, \n", "\n", " completion_only_loss = False,\n", " \n", " dataset_num_proc = dataset_nb_processors,\n", " \n", " max_steps = training_max_steps,\n", " num_train_epochs = training_num_train_epochs,\n", " warmup_ratio = 0.1,\n", "\n", " learning_rate = training_learning_rate,\n", "\n", " fp16 = not is_bfloat16_supported(),\n", " bf16 = is_bfloat16_supported(),\n", "\n", " optim = \"adamw_8bit\",\n", " weight_decay = 0.01,\n", " lr_scheduler_type = \"linear\",\n", " seed = seed,\n", "\n", " logging_steps = 1,\n", " eval_strategy = 'steps',\n", " eval_steps = 0.05, \n", " \n", " output_dir = \"outputs\",\n", " report_to = training_report_to,\n", " trackio_space_id = training_space_id,\n", " run_name = project_name.lower() + '-' + model_version + \"-sft\",\n", " ),\n", ")\n" ] }, { "cell_type": "markdown", "metadata": { "editable": true, "id": "WdoyA7uFXClH", "slideshow": { "slide_type": "" }, "tags": [] }, "source": [ "### Training Execution\n", "Execute the training process with the configured trainer and monitor the training progress." ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "2ejIt2xSNKKp" }, "outputs": [], "source": [ "gpu_stats = torch.cuda.get_device_properties(0)\n", "start_gpu_memory = round(torch.cuda.max_memory_reserved() / 1024 / 1024 / 1024, 3)\n", "max_memory = round(gpu_stats.total_memory / 1024 / 1024 / 1024, 3)\n", "print(f\"GPU = {gpu_stats.name}. Max memory = {max_memory} GB.\")\n", "print(f\"{start_gpu_memory} GB of memory reserved.\")" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "FTfullD_VNtG", "scrolled": true }, "outputs": [], "source": [ "trainer_stats = trainer.train()" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "pCqnaKmlO1U9" }, "outputs": [], "source": [ "used_memory = round(torch.cuda.max_memory_reserved() / 1024 / 1024 / 1024, 3)\n", "used_memory_for_lora = round(used_memory - start_gpu_memory, 3)\n", "used_percentage = round(used_memory /max_memory*100, 3)\n", "lora_percentage = round(used_memory_for_lora/max_memory*100, 3)\n", "print(f\"{trainer_stats.metrics['train_runtime']} seconds used for training.\")\n", "print(f\"{round(trainer_stats.metrics['train_runtime']/60, 2)} minutes used for training.\")\n", "print(f\"Peak reserved memory = {used_memory} GB.\")\n", "print(f\"Peak reserved memory for training = {used_memory_for_lora} GB.\")\n", "print(f\"Peak reserved memory % of max memory = {used_percentage} %.\")\n", "print(f\"Peak reserved memory for training % of max memory = {lora_percentage} %.\")" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## Save and test" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "saveModel(model_name + \"-sft\", ['q4_k_m'], hf_repo = project_name)\n", "testModel()" ] }, { "cell_type": "markdown", "metadata": { "id": "OL63RIybsz-V" }, "source": [ "# Direct Preference Optimization (DPO)" ] }, { "cell_type": "markdown", "metadata": { "editable": true, "id": "dwqnogVgtqxZ", "slideshow": { "slide_type": "" }, "tags": [] }, "source": [ "## Datasets" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "editable": true, "id": "96KmlYZJu9_L", "slideshow": { "slide_type": "" }, "tags": [] }, "outputs": [], "source": [ "#@title Init datasets\n", "datasets = {}" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "editable": true, "id": "CAFxrVKithB2", "slideshow": { "slide_type": "" }, "tags": [] }, "outputs": [], "source": [ "#@title French ORCA DPO\n", "datasets['aya'] = loadDataset('CATIE-AQ/aya_french_dpo', 'dpo')" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "editable": true, "id": "eglifcjNvI40", "slideshow": { "slide_type": "" }, "tags": [] }, "outputs": [], "source": [ "#@title Facebook community\n", "datasets['facebook'] = loadDataset('CATIE-AQ/facebook-community-alignment-dataset_french_dpo', 'dpo')" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "#@title Facebook Menlo\n", "datasets['menlo'] = loadDataset('CATIE-AQ/facebook_menlo_french_dpo', 'dpo')" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "editable": true, "id": "TaprZ-DJOhaK", "slideshow": { "slide_type": "" }, "tags": [] }, "outputs": [], "source": [ "#@title Dump datasets\n", "dumpDatasets(datasets)" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "editable": true, "id": "86ZOesSCvGd4", "slideshow": { "slide_type": "" }, "tags": [] }, "outputs": [], "source": [ "#@title Merge datasets\n", "dataset = mergeDatasets(datasets)" ] }, { "cell_type": "markdown", "metadata": { "editable": true, "id": "4ZOebQjKzbHZ", "slideshow": { "slide_type": "" }, "tags": [] }, "source": [ "## Training\n" ] }, { "cell_type": "markdown", "metadata": { "id": "kuJNJmP6zoU0" }, "source": [ "### Setup Trainer" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "editable": true, "id": "f9v8L5Z5zoyN", "slideshow": { "slide_type": "" }, "tags": [] }, "outputs": [], "source": [ "# Enable reward modelling stats\n", "from unsloth import PatchDPOTrainer\n", "PatchDPOTrainer()\n", "from trl import DPOTrainer, DPOConfig\n", "from unsloth import is_bfloat16_supported\n", "\n", "trainer = DPOTrainer(\n", " model = model,\n", " ref_model = None,\n", " beta = 0.1,\n", " train_dataset = dataset['train'],\n", " eval_dataset = dataset['test'],\n", " tokenizer = tokenizer,\n", " max_length = max_seq_length,\n", " max_prompt_length = max_seq_length,\n", " \n", " args = DPOConfig(\n", " per_device_train_batch_size = training_per_device_train_batch_size,\n", " gradient_accumulation_steps = training_gradient_accumulation_steps,\n", "\n", " dataset_num_proc = dataset_nb_processors,\n", " \n", " max_steps = training_max_steps,\n", " num_train_epochs = training_num_train_epochs,\n", " warmup_ratio = 0.1,\n", "\n", " learning_rate = training_learning_rate,\n", "\n", " fp16 = not is_bfloat16_supported(),\n", " bf16 = is_bfloat16_supported(),\n", "\n", " optim = \"adamw_8bit\",\n", " weight_decay = 0.01,\n", " lr_scheduler_type = \"linear\",\n", " seed = seed,\n", "\n", " logging_steps = 1,\n", " eval_strategy = 'steps',\n", " eval_steps = 0.05, \n", " \n", " output_dir = \"outputs\",\n", " report_to = training_report_to,\n", " trackio_space_id = training_space_id,\n", " run_name = project_name.lower() + '-' + model_version + \"-dpo\",\n", " ),\n", ")" ] }, { "cell_type": "markdown", "metadata": { "id": "MawSuz-HKR-q" }, "source": [ "### Training Execution\n", "Execute the training process with the configured trainer and monitor the training progress." ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "wD7PSLpezzB8" }, "outputs": [], "source": [ "gpu_stats = torch.cuda.get_device_properties(0)\n", "start_gpu_memory = round(torch.cuda.max_memory_reserved() / 1024 / 1024 / 1024, 3)\n", "max_memory = round(gpu_stats.total_memory / 1024 / 1024 / 1024, 3)\n", "print(f\"GPU = {gpu_stats.name}. Max memory = {max_memory} GB.\")\n", "print(f\"{start_gpu_memory} GB of memory reserved.\")" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "yqxqAZ7KJ4oL", "scrolled": true }, "outputs": [], "source": [ "trainer.train()" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "id": "E_70WjCozzB9" }, "outputs": [], "source": [ "used_memory = round(torch.cuda.max_memory_reserved() / 1024 / 1024 / 1024, 3)\n", "used_memory_for_lora = round(used_memory - start_gpu_memory, 3)\n", "used_percentage = round(used_memory /max_memory*100, 3)\n", "lora_percentage = round(used_memory_for_lora/max_memory*100, 3)\n", "print(f\"Peak reserved memory = {used_memory} GB.\")\n", "print(f\"Peak reserved memory for training = {used_memory_for_lora} GB.\")\n", "print(f\"Peak reserved memory % of max memory = {used_percentage} %.\")\n", "print(f\"Peak reserved memory for training % of max memory = {lora_percentage} %.\")" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "## Save and test" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [ "saveModel(model_name + \"-dpo\", ['q4_k_m'], hf_repo = project_name)\n", "saveModel(model_name, quantizations)\n", "testModel()" ] } ], "metadata": { "accelerator": "GPU", "colab": { "gpuType": "T4", "include_colab_link": true, "provenance": [] }, "kernelspec": { "display_name": "Python 3 (ipykernel)", "language": "python", "name": "python3" }, "language_info": { "codemirror_mode": { "name": "ipython", "version": 3 }, "file_extension": ".py", "mimetype": "text/x-python", "name": "python", "nbconvert_exporter": "python", "pygments_lexer": "ipython3", "version": "3.11.13" } }, "nbformat": 4, "nbformat_minor": 4 }