From 49da3ef739d4282572458dcdf377950abf28c2f6 Mon Sep 17 00:00:00 2001 From: edbeeching Date: Thu, 9 Nov 2023 10:56:25 +0100 Subject: [PATCH] adds configs and instructions for lora training --- recipes/zephyr-7b/README.md | 41 ++++++++++++++++++- recipes/zephyr-7b/dpo/config_full.yaml | 4 +- recipes/zephyr-7b/dpo/config_lora.yaml | 53 +++++++++++++++++++++++++ recipes/zephyr-7b/sft/config_full.yaml | 4 +- recipes/zephyr-7b/sft/config_lora.yaml | 54 ++++++++++++++++++++++++++ 5 files changed, 150 insertions(+), 6 deletions(-) create mode 100644 recipes/zephyr-7b/dpo/config_lora.yaml create mode 100644 recipes/zephyr-7b/sft/config_lora.yaml diff --git a/recipes/zephyr-7b/README.md b/recipes/zephyr-7b/README.md index cb3847a..f1df22c 100644 --- a/recipes/zephyr-7b/README.md +++ b/recipes/zephyr-7b/README.md @@ -1,7 +1,44 @@ +# Instructions +In the handbook, for each training step we provide two sets of recipes: +- Full training on a multi-GPU machine (tested on a 8xA100 node), using slurm to queue jobs. +- LORA taining on a single consumer 24GB GPU (tested on a RTX 4090) + +The full training jobs will scale to a multi-node setting, by adjusting `--nodes=1`, we advise adjusting the gradient accumulation steps and/or batch size if you want to replicate our results. + -## SFT +## Full training examples + +### SFT ```shell -sbatch --job-name=handbook_sft --nodes=1 recipes/launch.slurm zephyr-7b sft full +sbatch --job-name=handbook_sft --nodes=1 recipes/launch.slurm zephyr-7b sft full deepspeed_zero3 +``` + +## DPO +```shell +sbatch --job-name=handbook_sft --nodes=1 recipes/launch.slurm zephyr-7b sft full deepspeed_zero3 +``` + +## LORA training examples +### SFT +```shell +# locally on 1 gpu +accelerate launch scripts/run_sft.py recipes/zephyr-7b/sft/config_lora.yaml +``` + +```shell +# on a cluster +sbatch --job-name=handbook_sft_lora --nodes=1 recipes/launch.slurm zephyr-7b sft lora multi_gpu "--gradient_accumulation_steps=16" +``` + +### SFT +```shell +# locally on 1 gpu +accelerate launch scripts/run_dpo.py recipes/zephyr-7b/dpo/config_lora.yaml +``` + +```shell +# on a cluster +sbatch --job-name=handbook_dpo_lora --nodes=1 recipes/launch.slurm zephyr-7b dpo lora multi_gpu "--gradient_accumulation_steps=8" ``` \ No newline at end of file diff --git a/recipes/zephyr-7b/dpo/config_full.yaml b/recipes/zephyr-7b/dpo/config_full.yaml index 82258b8..8845330 100644 --- a/recipes/zephyr-7b/dpo/config_full.yaml +++ b/recipes/zephyr-7b/dpo/config_full.yaml @@ -18,7 +18,7 @@ evaluation_strategy: steps eval_steps: 100 gradient_accumulation_steps: 1 gradient_checkpointing: true -hub_model_id: zephyr-7b-dpo +hub_model_id: zephyr-7b-dpo-full learning_rate: 5.0e-7 log_level: info logging_steps: 10 @@ -27,7 +27,7 @@ max_length: 1024 max_prompt_length: 512 num_train_epochs: 3 optim: rmsprop -output_dir: data/zephyr-7b-dpo +output_dir: data/zephyr-7b-dpo-full per_device_train_batch_size: 4 per_device_eval_batch_size: 4 push_to_hub: true diff --git a/recipes/zephyr-7b/dpo/config_lora.yaml b/recipes/zephyr-7b/dpo/config_lora.yaml new file mode 100644 index 0000000..d3f00b9 --- /dev/null +++ b/recipes/zephyr-7b/dpo/config_lora.yaml @@ -0,0 +1,53 @@ +# Model arguments +model_name_or_path: HuggingFaceH4/mistral-7b-ift +model_revision: v14.0 +torch_dtype: auto + +# LORA +use_peft: true +lora_r: 64 +lora_alpha: 16 +lora_dropout: 0.1 +lora_target_modules: +- q_proj +- k_proj +- v_proj +- o_proj + +# Data training arguments + +dataset_mixer: + HuggingFaceH4/ultrafeedback_binarized: 1.0 +dataset_splits: +- train_prefs +- test_prefs +preprocessing_num_workers: 12 + +# DPOTrainer arguments +bf16: true +beta: 0.1 +do_eval: true +ddp_find_unused_parameters: true +evaluation_strategy: epoch +eval_steps: 100 +gradient_accumulation_steps: 16 +gradient_checkpointing: true +gradient_checkpointing_kwargs: + use_reentrant: False +hub_model_id: zephyr-7b-dpo-lora +learning_rate: 5.0e-7 +log_level: info +logging_steps: 10 +lr_scheduler_type: linear +max_length: 1024 +max_prompt_length: 512 +num_train_epochs: 3 +optim: rmsprop +output_dir: data/zephyr-7b-dpo-lora # It is handy to append `hub_model_revision` to keep track of your local experiments +per_device_train_batch_size: 2 +per_device_eval_batch_size: 4 +push_to_hub: true +save_strategy: "no" +save_total_limit: null +seed: 42 +warmup_ratio: 0.1 \ No newline at end of file diff --git a/recipes/zephyr-7b/sft/config_full.yaml b/recipes/zephyr-7b/sft/config_full.yaml index 8ceb856..020df7f 100644 --- a/recipes/zephyr-7b/sft/config_full.yaml +++ b/recipes/zephyr-7b/sft/config_full.yaml @@ -17,7 +17,7 @@ bf16: true evaluation_strategy: epoch gradient_accumulation_steps: 2 gradient_checkpointing: true -hub_model_id: zephyr-7b-sft +hub_model_id: zephyr-7b-sft-full hub_strategy: every_save learning_rate: 2.0e-05 log_level: info @@ -27,7 +27,7 @@ lr_scheduler_type: cosine max_seq_length: 2048 max_steps: -1 num_train_epochs: 1 -output_dir: data/zephyr-7b-sft +output_dir: data/zephyr-7b-sft-full overwrite_output_dir: true per_device_eval_batch_size: 16 per_device_train_batch_size: 32 diff --git a/recipes/zephyr-7b/sft/config_lora.yaml b/recipes/zephyr-7b/sft/config_lora.yaml new file mode 100644 index 0000000..6bf806b --- /dev/null +++ b/recipes/zephyr-7b/sft/config_lora.yaml @@ -0,0 +1,54 @@ +# Model arguments +model_name_or_path: mistralai/Mistral-7B-v0.1 +model_revision: main +torch_dtype: auto +use_flash_attention_2: true + +# LORA +use_peft: true +lora_r: 64 +lora_alpha: 16 +lora_dropout: 0.1 +lora_target_modules: +- q_proj +- k_proj +- v_proj +- o_proj + +# Data training arguments +dataset_mixer: + HuggingFaceH4/ultrachat_200k: 1.0 +dataset_splits: +- train_sft +- test_sft +preprocessing_num_workers: 12 + +# SFT trainer config +bf16: true +evaluation_strategy: epoch +gradient_accumulation_steps: 128 +ddp_find_unused_parameters: true +gradient_checkpointing: true +gradient_checkpointing_kwargs: + use_reentrant: False +hub_model_id: zephyr-7b-sft-lora +hub_strategy: every_save +learning_rate: 2.0e-05 +log_level: info +logging_steps: 5 +logging_strategy: steps +lr_scheduler_type: cosine +max_seq_length: 2048 +max_steps: -1 +num_train_epochs: 1 +output_dir: data/zephyr-7b-sft-lora +overwrite_output_dir: true +per_device_eval_batch_size: 8 +per_device_train_batch_size: 4 +push_to_hub: True +report_to: +- tensorboard +save_strategy: "no" +save_total_limit: null +seed: 42 +tf32: true \ No newline at end of file