Instructions to use xuewei-huang/peft-llm-study-adapters with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- PEFT
How to use xuewei-huang/peft-llm-study-adapters with PEFT:
Task type is invalid.
- Notebooks
- Google Colab
- Kaggle
Add adapter index and metadata
Browse files- README.md +52 -0
- adapter_index.json +272 -0
- metadata/lora_starcoder2_alpaca/args.json +40 -0
- metadata/lora_starcoder2_alpaca/metrics.json +36 -0
- metadata/mistral7b_lora_alpaca/args.json +41 -0
- metadata/mistral7b_lora_alpaca/metrics.json +36 -0
- metadata/mistral7b_lora_alpaca_cot/args.json +41 -0
- metadata/mistral7b_lora_alpaca_cot/metrics.json +36 -0
- metadata/mistral7b_lora_guanaco/args.json +41 -0
- metadata/mistral7b_lora_guanaco/metrics.json +36 -0
- metadata/mistral7b_prompt_alpaca_cot/args.json +33 -0
- metadata/mistral7b_prompt_alpaca_cot/metrics.json +36 -0
- metadata/mistral7b_prompt_guanaco/args.json +33 -0
- metadata/mistral7b_prompt_guanaco/metrics.json +36 -0
- metadata/mistral7b_prompt_tuning_alpaca/args.json +33 -0
- metadata/mistral7b_prompt_tuning_alpaca/metrics.json +36 -0
- metadata/mistral7b_qlora_alpaca/args.json +41 -0
- metadata/mistral7b_qlora_alpaca/metrics.json +36 -0
- metadata/mistral7b_qlora_alpaca_cot/args.json +41 -0
- metadata/mistral7b_qlora_alpaca_cot/metrics.json +36 -0
- metadata/mistral7b_qlora_guanaco/args.json +41 -0
- metadata/mistral7b_qlora_guanaco/metrics.json +36 -0
- metadata/prompt_tuning_starcoder2_alpaca/args.json +33 -0
- metadata/prompt_tuning_starcoder2_alpaca/metrics.json +36 -0
- metadata/qlora_starcoder2_alpaca/args.json +40 -0
- metadata/qlora_starcoder2_alpaca/metrics.json +36 -0
- metadata/starcoder2_lora_alpaca_cot/args.json +40 -0
- metadata/starcoder2_lora_alpaca_cot/metrics.json +36 -0
- metadata/starcoder2_lora_guanaco/args.json +40 -0
- metadata/starcoder2_lora_guanaco/metrics.json +36 -0
- metadata/starcoder2_prompt_alpaca_cot/args.json +33 -0
- metadata/starcoder2_prompt_alpaca_cot/metrics.json +36 -0
- metadata/starcoder2_prompt_guanaco/args.json +33 -0
- metadata/starcoder2_prompt_guanaco/metrics.json +36 -0
- metadata/starcoder2_qlora_alpaca_cot/args.json +40 -0
- metadata/starcoder2_qlora_alpaca_cot/metrics.json +36 -0
- metadata/starcoder2_qlora_guanaco/args.json +40 -0
- metadata/starcoder2_qlora_guanaco/metrics.json +36 -0
README.md
ADDED
|
@@ -0,0 +1,52 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
library_name: peft
|
| 3 |
+
tags:
|
| 4 |
+
- peft
|
| 5 |
+
- lora
|
| 6 |
+
- qlora
|
| 7 |
+
- prompt-tuning
|
| 8 |
+
- instruction-tuning
|
| 9 |
+
---
|
| 10 |
+
|
| 11 |
+
# PEFT LLM Study Adapters
|
| 12 |
+
|
| 13 |
+
This repository stores PEFT adapter weights only. Base model weights are not included.
|
| 14 |
+
|
| 15 |
+
Load a specific adapter from its subfolder, for example:
|
| 16 |
+
|
| 17 |
+
```python
|
| 18 |
+
from transformers import AutoModelForCausalLM
|
| 19 |
+
from peft import PeftModel
|
| 20 |
+
|
| 21 |
+
base = AutoModelForCausalLM.from_pretrained("bigcode/starcoder2-3b")
|
| 22 |
+
model = PeftModel.from_pretrained(base, "REPO_ID", subfolder="adapters/RUN_NAME")
|
| 23 |
+
```
|
| 24 |
+
|
| 25 |
+
## Adapter Index
|
| 26 |
+
|
| 27 |
+
| Run | Method | Base Model | Dataset | Steps | Eval Loss | PPL | Peak GPU GB | Path |
|
| 28 |
+
| --- | --- | --- | --- | ---: | ---: | ---: | ---: | --- |
|
| 29 |
+
| lora_starcoder2_alpaca | lora | bigcode/starcoder2-3b | tatsu-lab/alpaca | 500 | 1.4354 | 4.2015 | 6.4372 | adapters/lora_starcoder2_alpaca |
|
| 30 |
+
| mistral7b_lora_alpaca | lora | mistralai/Mistral-7B-v0.1 | tatsu-lab/alpaca | 500 | 1.6779 | 5.3541 | 14.4979 | adapters/mistral7b_lora_alpaca |
|
| 31 |
+
| mistral7b_lora_alpaca_cot | lora | mistralai/Mistral-7B-v0.1 | QingyiSi/Alpaca-CoT/combination/alcapa_plus_cot.json | 500 | 1.5490 | 4.7067 | 14.4976 | adapters/mistral7b_lora_alpaca_cot |
|
| 32 |
+
| mistral7b_lora_guanaco | lora | mistralai/Mistral-7B-v0.1 | fengtc/GuanacoDataset | 500 | 1.4493 | 4.2602 | 14.4984 | adapters/mistral7b_lora_guanaco |
|
| 33 |
+
| mistral7b_prompt_alpaca_cot | prompt_tuning | mistralai/Mistral-7B-v0.1 | QingyiSi/Alpaca-CoT/combination/alcapa_plus_cot.json | 500 | 1.0121 | 2.7515 | 13.8476 | adapters/mistral7b_prompt_alpaca_cot |
|
| 34 |
+
| mistral7b_prompt_guanaco | prompt_tuning | mistralai/Mistral-7B-v0.1 | fengtc/GuanacoDataset | 500 | 1.0030 | 2.7266 | 13.8476 | adapters/mistral7b_prompt_guanaco |
|
| 35 |
+
| mistral7b_prompt_tuning_alpaca | prompt_tuning | mistralai/Mistral-7B-v0.1 | tatsu-lab/alpaca | 500 | 1.0024 | 2.7247 | 13.8487 | adapters/mistral7b_prompt_tuning_alpaca |
|
| 36 |
+
| mistral7b_qlora_alpaca | qlora | mistralai/Mistral-7B-v0.1 | tatsu-lab/alpaca | 500 | 1.7437 | 5.7184 | 5.7184 | adapters/mistral7b_qlora_alpaca |
|
| 37 |
+
| mistral7b_qlora_alpaca_cot | qlora | mistralai/Mistral-7B-v0.1 | QingyiSi/Alpaca-CoT/combination/alcapa_plus_cot.json | 500 | 1.5379 | 4.6548 | 5.7184 | adapters/mistral7b_qlora_alpaca_cot |
|
| 38 |
+
| mistral7b_qlora_guanaco | qlora | mistralai/Mistral-7B-v0.1 | fengtc/GuanacoDataset | 500 | 1.4491 | 4.2591 | 5.7177 | adapters/mistral7b_qlora_guanaco |
|
| 39 |
+
| prompt_tuning_starcoder2_alpaca | prompt_tuning | bigcode/starcoder2-3b | tatsu-lab/alpaca | 500 | 1.5602 | 4.7596 | 6.1643 | adapters/prompt_tuning_starcoder2_alpaca |
|
| 40 |
+
| qlora_starcoder2_alpaca | qlora | bigcode/starcoder2-3b | tatsu-lab/alpaca | 500 | 1.4461 | 4.2465 | 3.0791 | adapters/qlora_starcoder2_alpaca |
|
| 41 |
+
| starcoder2_lora_alpaca_cot | lora | bigcode/starcoder2-3b | QingyiSi/Alpaca-CoT/combination/alcapa_plus_cot.json | 500 | 1.2603 | 3.5264 | 6.4582 | adapters/starcoder2_lora_alpaca_cot |
|
| 42 |
+
| starcoder2_lora_guanaco | lora | bigcode/starcoder2-3b | fengtc/GuanacoDataset | 500 | 1.2887 | 3.6282 | 6.4578 | adapters/starcoder2_lora_guanaco |
|
| 43 |
+
| starcoder2_prompt_alpaca_cot | prompt_tuning | bigcode/starcoder2-3b | QingyiSi/Alpaca-CoT/combination/alcapa_plus_cot.json | 500 | 1.6412 | 5.1614 | 6.1846 | adapters/starcoder2_prompt_alpaca_cot |
|
| 44 |
+
| starcoder2_prompt_guanaco | prompt_tuning | bigcode/starcoder2-3b | fengtc/GuanacoDataset | 500 | 1.3930 | 4.0268 | 6.1854 | adapters/starcoder2_prompt_guanaco |
|
| 45 |
+
| starcoder2_qlora_alpaca_cot | qlora | bigcode/starcoder2-3b | QingyiSi/Alpaca-CoT/combination/alcapa_plus_cot.json | 500 | 1.2595 | 3.5237 | 3.0788 | adapters/starcoder2_qlora_alpaca_cot |
|
| 46 |
+
| starcoder2_qlora_guanaco | qlora | bigcode/starcoder2-3b | fengtc/GuanacoDataset | 500 | 1.2983 | 3.6632 | 3.0791 | adapters/starcoder2_qlora_guanaco |
|
| 47 |
+
|
| 48 |
+
## Notes
|
| 49 |
+
|
| 50 |
+
- These are adapters trained for an empirical PEFT comparison project.
|
| 51 |
+
- Guanaco proxy runs use `fengtc/GuanacoDataset` when the originally listed gated dataset is unavailable.
|
| 52 |
+
- See `adapter_index.json` and `metadata/*/metrics.json` for machine-readable details.
|
adapter_index.json
ADDED
|
@@ -0,0 +1,272 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[
|
| 2 |
+
{
|
| 3 |
+
"base_model": "bigcode/starcoder2-3b",
|
| 4 |
+
"dataset": "tatsu-lab/alpaca",
|
| 5 |
+
"dataset_file": null,
|
| 6 |
+
"eval_loss": 1.4354339838027954,
|
| 7 |
+
"max_steps": 500,
|
| 8 |
+
"method": "lora",
|
| 9 |
+
"path_in_repo": "adapters/lora_starcoder2_alpaca",
|
| 10 |
+
"peak_gpu_memory_gb": 6.437230587005615,
|
| 11 |
+
"perplexity": 4.201467982241092,
|
| 12 |
+
"run": "lora_starcoder2_alpaca",
|
| 13 |
+
"train_loss": 0.7681699752807617,
|
| 14 |
+
"trainable_params": 23838720,
|
| 15 |
+
"trainable_percent": 0.7805199912694414
|
| 16 |
+
},
|
| 17 |
+
{
|
| 18 |
+
"base_model": "mistralai/Mistral-7B-v0.1",
|
| 19 |
+
"dataset": "tatsu-lab/alpaca",
|
| 20 |
+
"dataset_file": null,
|
| 21 |
+
"eval_loss": 1.6778709888458252,
|
| 22 |
+
"max_steps": 500,
|
| 23 |
+
"method": "lora",
|
| 24 |
+
"path_in_repo": "adapters/mistral7b_lora_alpaca",
|
| 25 |
+
"peak_gpu_memory_gb": 14.497857093811035,
|
| 26 |
+
"perplexity": 5.354144794182875,
|
| 27 |
+
"run": "mistral7b_lora_alpaca",
|
| 28 |
+
"train_loss": 0.24137365317344667,
|
| 29 |
+
"trainable_params": 41943040,
|
| 30 |
+
"trainable_percent": 0.5758499550960753
|
| 31 |
+
},
|
| 32 |
+
{
|
| 33 |
+
"base_model": "mistralai/Mistral-7B-v0.1",
|
| 34 |
+
"dataset": "QingyiSi/Alpaca-CoT",
|
| 35 |
+
"dataset_file": "combination/alcapa_plus_cot.json",
|
| 36 |
+
"eval_loss": 1.5489877462387085,
|
| 37 |
+
"max_steps": 500,
|
| 38 |
+
"method": "lora",
|
| 39 |
+
"path_in_repo": "adapters/mistral7b_lora_alpaca_cot",
|
| 40 |
+
"peak_gpu_memory_gb": 14.497621536254883,
|
| 41 |
+
"perplexity": 4.706703392184987,
|
| 42 |
+
"run": "mistral7b_lora_alpaca_cot",
|
| 43 |
+
"train_loss": 0.4384129907488823,
|
| 44 |
+
"trainable_params": 41943040,
|
| 45 |
+
"trainable_percent": 0.5758499550960753
|
| 46 |
+
},
|
| 47 |
+
{
|
| 48 |
+
"base_model": "mistralai/Mistral-7B-v0.1",
|
| 49 |
+
"dataset": "fengtc/GuanacoDataset",
|
| 50 |
+
"dataset_file": null,
|
| 51 |
+
"eval_loss": 1.4493114948272705,
|
| 52 |
+
"max_steps": 500,
|
| 53 |
+
"method": "lora",
|
| 54 |
+
"path_in_repo": "adapters/mistral7b_lora_guanaco",
|
| 55 |
+
"peak_gpu_memory_gb": 14.498394012451172,
|
| 56 |
+
"perplexity": 4.2601803489833925,
|
| 57 |
+
"run": "mistral7b_lora_guanaco",
|
| 58 |
+
"train_loss": 0.5458194477558136,
|
| 59 |
+
"trainable_params": 41943040,
|
| 60 |
+
"trainable_percent": 0.5758499550960753
|
| 61 |
+
},
|
| 62 |
+
{
|
| 63 |
+
"base_model": "mistralai/Mistral-7B-v0.1",
|
| 64 |
+
"dataset": "QingyiSi/Alpaca-CoT",
|
| 65 |
+
"dataset_file": "combination/alcapa_plus_cot.json",
|
| 66 |
+
"eval_loss": 1.012134313583374,
|
| 67 |
+
"max_steps": 500,
|
| 68 |
+
"method": "prompt_tuning",
|
| 69 |
+
"path_in_repo": "adapters/mistral7b_prompt_alpaca_cot",
|
| 70 |
+
"peak_gpu_memory_gb": 13.847609996795654,
|
| 71 |
+
"perplexity": 2.7514672465197147,
|
| 72 |
+
"run": "mistral7b_prompt_alpaca_cot",
|
| 73 |
+
"train_loss": 1.0451534767150878,
|
| 74 |
+
"trainable_params": 131072,
|
| 75 |
+
"trainable_percent": 0.0018099209686697024
|
| 76 |
+
},
|
| 77 |
+
{
|
| 78 |
+
"base_model": "mistralai/Mistral-7B-v0.1",
|
| 79 |
+
"dataset": "fengtc/GuanacoDataset",
|
| 80 |
+
"dataset_file": null,
|
| 81 |
+
"eval_loss": 1.0030479431152344,
|
| 82 |
+
"max_steps": 500,
|
| 83 |
+
"method": "prompt_tuning",
|
| 84 |
+
"path_in_repo": "adapters/mistral7b_prompt_guanaco",
|
| 85 |
+
"peak_gpu_memory_gb": 13.847609519958496,
|
| 86 |
+
"perplexity": 2.7265796360422554,
|
| 87 |
+
"run": "mistral7b_prompt_guanaco",
|
| 88 |
+
"train_loss": 1.0741594829559327,
|
| 89 |
+
"trainable_params": 131072,
|
| 90 |
+
"trainable_percent": 0.0018099209686697024
|
| 91 |
+
},
|
| 92 |
+
{
|
| 93 |
+
"base_model": "mistralai/Mistral-7B-v0.1",
|
| 94 |
+
"dataset": "tatsu-lab/alpaca",
|
| 95 |
+
"dataset_file": null,
|
| 96 |
+
"eval_loss": 1.0023709535598755,
|
| 97 |
+
"max_steps": 500,
|
| 98 |
+
"method": "prompt_tuning",
|
| 99 |
+
"path_in_repo": "adapters/mistral7b_prompt_tuning_alpaca",
|
| 100 |
+
"peak_gpu_memory_gb": 13.848701477050781,
|
| 101 |
+
"perplexity": 2.724734394781806,
|
| 102 |
+
"run": "mistral7b_prompt_tuning_alpaca",
|
| 103 |
+
"train_loss": 0.5351699771881103,
|
| 104 |
+
"trainable_params": 131072,
|
| 105 |
+
"trainable_percent": 0.0018099209686697024
|
| 106 |
+
},
|
| 107 |
+
{
|
| 108 |
+
"base_model": "mistralai/Mistral-7B-v0.1",
|
| 109 |
+
"dataset": "tatsu-lab/alpaca",
|
| 110 |
+
"dataset_file": null,
|
| 111 |
+
"eval_loss": 1.743696689605713,
|
| 112 |
+
"max_steps": 500,
|
| 113 |
+
"method": "qlora",
|
| 114 |
+
"path_in_repo": "adapters/mistral7b_qlora_alpaca",
|
| 115 |
+
"peak_gpu_memory_gb": 5.718385696411133,
|
| 116 |
+
"perplexity": 5.718443709459332,
|
| 117 |
+
"run": "mistral7b_qlora_alpaca",
|
| 118 |
+
"train_loss": 0.5023761793375016,
|
| 119 |
+
"trainable_params": 41943040,
|
| 120 |
+
"trainable_percent": 1.1055056122762943
|
| 121 |
+
},
|
| 122 |
+
{
|
| 123 |
+
"base_model": "mistralai/Mistral-7B-v0.1",
|
| 124 |
+
"dataset": "QingyiSi/Alpaca-CoT",
|
| 125 |
+
"dataset_file": "combination/alcapa_plus_cot.json",
|
| 126 |
+
"eval_loss": 1.5378942489624023,
|
| 127 |
+
"max_steps": 500,
|
| 128 |
+
"method": "qlora",
|
| 129 |
+
"path_in_repo": "adapters/mistral7b_qlora_alpaca_cot",
|
| 130 |
+
"peak_gpu_memory_gb": 5.718398094177246,
|
| 131 |
+
"perplexity": 4.654778139763514,
|
| 132 |
+
"run": "mistral7b_qlora_alpaca_cot",
|
| 133 |
+
"train_loss": 0.4474883066415787,
|
| 134 |
+
"trainable_params": 41943040,
|
| 135 |
+
"trainable_percent": 1.1055056122762943
|
| 136 |
+
},
|
| 137 |
+
{
|
| 138 |
+
"base_model": "mistralai/Mistral-7B-v0.1",
|
| 139 |
+
"dataset": "fengtc/GuanacoDataset",
|
| 140 |
+
"dataset_file": null,
|
| 141 |
+
"eval_loss": 1.4490643739700317,
|
| 142 |
+
"max_steps": 500,
|
| 143 |
+
"method": "qlora",
|
| 144 |
+
"path_in_repo": "adapters/mistral7b_qlora_guanaco",
|
| 145 |
+
"peak_gpu_memory_gb": 5.7177019119262695,
|
| 146 |
+
"perplexity": 4.259127699634722,
|
| 147 |
+
"run": "mistral7b_qlora_guanaco",
|
| 148 |
+
"train_loss": 0.5492446699142456,
|
| 149 |
+
"trainable_params": 41943040,
|
| 150 |
+
"trainable_percent": 1.1055056122762943
|
| 151 |
+
},
|
| 152 |
+
{
|
| 153 |
+
"base_model": "bigcode/starcoder2-3b",
|
| 154 |
+
"dataset": "tatsu-lab/alpaca",
|
| 155 |
+
"dataset_file": null,
|
| 156 |
+
"eval_loss": 1.5601716041564941,
|
| 157 |
+
"max_steps": 500,
|
| 158 |
+
"method": "prompt_tuning",
|
| 159 |
+
"path_in_repo": "adapters/prompt_tuning_starcoder2_alpaca",
|
| 160 |
+
"peak_gpu_memory_gb": 6.164307594299316,
|
| 161 |
+
"perplexity": 4.759637948716392,
|
| 162 |
+
"run": "prompt_tuning_starcoder2_alpaca",
|
| 163 |
+
"train_loss": 1.4749931907653808,
|
| 164 |
+
"trainable_params": 98304,
|
| 165 |
+
"trainable_percent": 0.0032438536575970546
|
| 166 |
+
},
|
| 167 |
+
{
|
| 168 |
+
"base_model": "bigcode/starcoder2-3b",
|
| 169 |
+
"dataset": "tatsu-lab/alpaca",
|
| 170 |
+
"dataset_file": null,
|
| 171 |
+
"eval_loss": 1.4460986852645874,
|
| 172 |
+
"max_steps": 500,
|
| 173 |
+
"method": "qlora",
|
| 174 |
+
"path_in_repo": "adapters/qlora_starcoder2_alpaca",
|
| 175 |
+
"peak_gpu_memory_gb": 3.079056739807129,
|
| 176 |
+
"perplexity": 4.2465151643917896,
|
| 177 |
+
"run": "qlora_starcoder2_alpaca",
|
| 178 |
+
"train_loss": 0.7822464866638184,
|
| 179 |
+
"trainable_params": 23838720,
|
| 180 |
+
"trainable_percent": 1.4760456432877014
|
| 181 |
+
},
|
| 182 |
+
{
|
| 183 |
+
"base_model": "bigcode/starcoder2-3b",
|
| 184 |
+
"dataset": "QingyiSi/Alpaca-CoT",
|
| 185 |
+
"dataset_file": "combination/alcapa_plus_cot.json",
|
| 186 |
+
"eval_loss": 1.2602890729904175,
|
| 187 |
+
"max_steps": 500,
|
| 188 |
+
"method": "lora",
|
| 189 |
+
"path_in_repo": "adapters/starcoder2_lora_alpaca_cot",
|
| 190 |
+
"peak_gpu_memory_gb": 6.45817756652832,
|
| 191 |
+
"perplexity": 3.5264407388091508,
|
| 192 |
+
"run": "starcoder2_lora_alpaca_cot",
|
| 193 |
+
"train_loss": 1.4059115772247315,
|
| 194 |
+
"trainable_params": 23838720,
|
| 195 |
+
"trainable_percent": 0.7805199912694414
|
| 196 |
+
},
|
| 197 |
+
{
|
| 198 |
+
"base_model": "bigcode/starcoder2-3b",
|
| 199 |
+
"dataset": "fengtc/GuanacoDataset",
|
| 200 |
+
"dataset_file": null,
|
| 201 |
+
"eval_loss": 1.2887365818023682,
|
| 202 |
+
"max_steps": 500,
|
| 203 |
+
"method": "lora",
|
| 204 |
+
"path_in_repo": "adapters/starcoder2_lora_guanaco",
|
| 205 |
+
"peak_gpu_memory_gb": 6.45782995223999,
|
| 206 |
+
"perplexity": 3.6281997252628484,
|
| 207 |
+
"run": "starcoder2_lora_guanaco",
|
| 208 |
+
"train_loss": 1.6917072792053223,
|
| 209 |
+
"trainable_params": 23838720,
|
| 210 |
+
"trainable_percent": 0.7805199912694414
|
| 211 |
+
},
|
| 212 |
+
{
|
| 213 |
+
"base_model": "bigcode/starcoder2-3b",
|
| 214 |
+
"dataset": "QingyiSi/Alpaca-CoT",
|
| 215 |
+
"dataset_file": "combination/alcapa_plus_cot.json",
|
| 216 |
+
"eval_loss": 1.6411982774734497,
|
| 217 |
+
"max_steps": 500,
|
| 218 |
+
"method": "prompt_tuning",
|
| 219 |
+
"path_in_repo": "adapters/starcoder2_prompt_alpaca_cot",
|
| 220 |
+
"peak_gpu_memory_gb": 6.1845526695251465,
|
| 221 |
+
"perplexity": 5.161350538285551,
|
| 222 |
+
"run": "starcoder2_prompt_alpaca_cot",
|
| 223 |
+
"train_loss": 2.839910514831543,
|
| 224 |
+
"trainable_params": 98304,
|
| 225 |
+
"trainable_percent": 0.0032438536575970546
|
| 226 |
+
},
|
| 227 |
+
{
|
| 228 |
+
"base_model": "bigcode/starcoder2-3b",
|
| 229 |
+
"dataset": "fengtc/GuanacoDataset",
|
| 230 |
+
"dataset_file": null,
|
| 231 |
+
"eval_loss": 1.3929762840270996,
|
| 232 |
+
"max_steps": 500,
|
| 233 |
+
"method": "prompt_tuning",
|
| 234 |
+
"path_in_repo": "adapters/starcoder2_prompt_guanaco",
|
| 235 |
+
"peak_gpu_memory_gb": 6.185425758361816,
|
| 236 |
+
"perplexity": 4.026817187039079,
|
| 237 |
+
"run": "starcoder2_prompt_guanaco",
|
| 238 |
+
"train_loss": 3.4105590629577636,
|
| 239 |
+
"trainable_params": 98304,
|
| 240 |
+
"trainable_percent": 0.0032438536575970546
|
| 241 |
+
},
|
| 242 |
+
{
|
| 243 |
+
"base_model": "bigcode/starcoder2-3b",
|
| 244 |
+
"dataset": "QingyiSi/Alpaca-CoT",
|
| 245 |
+
"dataset_file": "combination/alcapa_plus_cot.json",
|
| 246 |
+
"eval_loss": 1.2594980001449585,
|
| 247 |
+
"max_steps": 500,
|
| 248 |
+
"method": "qlora",
|
| 249 |
+
"path_in_repo": "adapters/starcoder2_qlora_alpaca_cot",
|
| 250 |
+
"peak_gpu_memory_gb": 3.0788111686706543,
|
| 251 |
+
"perplexity": 3.5236521704253443,
|
| 252 |
+
"run": "starcoder2_qlora_alpaca_cot",
|
| 253 |
+
"train_loss": 0.6505789184570312,
|
| 254 |
+
"trainable_params": 23838720,
|
| 255 |
+
"trainable_percent": 1.4760456432877014
|
| 256 |
+
},
|
| 257 |
+
{
|
| 258 |
+
"base_model": "bigcode/starcoder2-3b",
|
| 259 |
+
"dataset": "fengtc/GuanacoDataset",
|
| 260 |
+
"dataset_file": null,
|
| 261 |
+
"eval_loss": 1.2983253002166748,
|
| 262 |
+
"max_steps": 500,
|
| 263 |
+
"method": "qlora",
|
| 264 |
+
"path_in_repo": "adapters/starcoder2_qlora_guanaco",
|
| 265 |
+
"peak_gpu_memory_gb": 3.079078197479248,
|
| 266 |
+
"perplexity": 3.6631568399040884,
|
| 267 |
+
"run": "starcoder2_qlora_guanaco",
|
| 268 |
+
"train_loss": 1.715220178604126,
|
| 269 |
+
"trainable_params": 23838720,
|
| 270 |
+
"trainable_percent": 1.4760456432877014
|
| 271 |
+
}
|
| 272 |
+
]
|
metadata/lora_starcoder2_alpaca/args.json
ADDED
|
@@ -0,0 +1,40 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"bf16": true,
|
| 3 |
+
"config": "configs/lora.yaml",
|
| 4 |
+
"dataset_file": null,
|
| 5 |
+
"dataset_name": "tatsu-lab/alpaca",
|
| 6 |
+
"dataset_split": "train",
|
| 7 |
+
"eval_steps": 50,
|
| 8 |
+
"gradient_accumulation_steps": 16,
|
| 9 |
+
"gradient_checkpointing": true,
|
| 10 |
+
"learning_rate": 0.0002,
|
| 11 |
+
"logging_steps": 10,
|
| 12 |
+
"lora_alpha": 32,
|
| 13 |
+
"lora_dropout": 0.05,
|
| 14 |
+
"lora_r": 16,
|
| 15 |
+
"max_eval_samples": 256,
|
| 16 |
+
"max_seq_length": 512,
|
| 17 |
+
"max_steps": 500,
|
| 18 |
+
"max_train_samples": 2000,
|
| 19 |
+
"method": "lora",
|
| 20 |
+
"model_name": "bigcode/starcoder2-3b",
|
| 21 |
+
"num_train_epochs": 1,
|
| 22 |
+
"num_virtual_tokens": 32,
|
| 23 |
+
"output_dir": "outputs/lora_starcoder2_alpaca",
|
| 24 |
+
"per_device_eval_batch_size": 2,
|
| 25 |
+
"per_device_train_batch_size": 1,
|
| 26 |
+
"prompt_tuning_init_text": "Follow the instruction carefully and provide a helpful, accurate response.",
|
| 27 |
+
"resume_from_checkpoint": "outputs/lora_starcoder2_alpaca/checkpoint-250",
|
| 28 |
+
"save_steps": 250,
|
| 29 |
+
"seed": 42,
|
| 30 |
+
"target_modules": [
|
| 31 |
+
"q_proj",
|
| 32 |
+
"k_proj",
|
| 33 |
+
"v_proj",
|
| 34 |
+
"o_proj",
|
| 35 |
+
"c_fc",
|
| 36 |
+
"c_proj"
|
| 37 |
+
],
|
| 38 |
+
"warmup_ratio": 0.03,
|
| 39 |
+
"weight_decay": 0.0
|
| 40 |
+
}
|
metadata/lora_starcoder2_alpaca/metrics.json
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_file": null,
|
| 3 |
+
"dataset_name": "tatsu-lab/alpaca",
|
| 4 |
+
"effective_batch_size": 16,
|
| 5 |
+
"eval_metrics": {
|
| 6 |
+
"epoch": 4.0,
|
| 7 |
+
"eval_loss": 1.4354339838027954,
|
| 8 |
+
"eval_runtime": 7.5045,
|
| 9 |
+
"eval_samples_per_second": 34.113,
|
| 10 |
+
"eval_steps_per_second": 17.057,
|
| 11 |
+
"perplexity": 4.201467982241092
|
| 12 |
+
},
|
| 13 |
+
"learning_rate": 0.0002,
|
| 14 |
+
"max_eval_samples": 256,
|
| 15 |
+
"max_seq_length": 512,
|
| 16 |
+
"max_steps": 500,
|
| 17 |
+
"max_train_samples": 2000,
|
| 18 |
+
"method": "lora",
|
| 19 |
+
"model_name": "bigcode/starcoder2-3b",
|
| 20 |
+
"parameter_counts": {
|
| 21 |
+
"total_params": 3054210048,
|
| 22 |
+
"trainable_params": 23838720,
|
| 23 |
+
"trainable_percent": 0.7805199912694414
|
| 24 |
+
},
|
| 25 |
+
"peak_gpu_memory_gb": 6.437230587005615,
|
| 26 |
+
"seed": 42,
|
| 27 |
+
"train_metrics": {
|
| 28 |
+
"epoch": 4.0,
|
| 29 |
+
"total_flos": 1.7099333093818368e+16,
|
| 30 |
+
"train_loss": 0.7681699752807617,
|
| 31 |
+
"train_runtime": 1315.7845,
|
| 32 |
+
"train_samples_per_second": 6.08,
|
| 33 |
+
"train_steps_per_second": 0.38
|
| 34 |
+
},
|
| 35 |
+
"wall_clock_seconds": 1316.437697172165
|
| 36 |
+
}
|
metadata/mistral7b_lora_alpaca/args.json
ADDED
|
@@ -0,0 +1,41 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"bf16": true,
|
| 3 |
+
"config": "configs/mistral_lora_alpaca.yaml",
|
| 4 |
+
"dataset_file": null,
|
| 5 |
+
"dataset_name": "tatsu-lab/alpaca",
|
| 6 |
+
"dataset_split": "train",
|
| 7 |
+
"eval_steps": 50,
|
| 8 |
+
"gradient_accumulation_steps": 16,
|
| 9 |
+
"gradient_checkpointing": true,
|
| 10 |
+
"learning_rate": 0.0002,
|
| 11 |
+
"logging_steps": 10,
|
| 12 |
+
"lora_alpha": 32,
|
| 13 |
+
"lora_dropout": 0.05,
|
| 14 |
+
"lora_r": 16,
|
| 15 |
+
"max_eval_samples": 256,
|
| 16 |
+
"max_seq_length": 512,
|
| 17 |
+
"max_steps": 500,
|
| 18 |
+
"max_train_samples": 2000,
|
| 19 |
+
"method": "lora",
|
| 20 |
+
"model_name": "mistralai/Mistral-7B-v0.1",
|
| 21 |
+
"num_train_epochs": 1,
|
| 22 |
+
"num_virtual_tokens": 32,
|
| 23 |
+
"output_dir": "outputs/mistral7b_lora_alpaca",
|
| 24 |
+
"per_device_eval_batch_size": 2,
|
| 25 |
+
"per_device_train_batch_size": 1,
|
| 26 |
+
"prompt_tuning_init_text": "Follow the instruction carefully and provide a helpful, accurate response.",
|
| 27 |
+
"resume_from_checkpoint": "outputs/mistral7b_lora_alpaca/checkpoint-125",
|
| 28 |
+
"save_steps": 250,
|
| 29 |
+
"seed": 42,
|
| 30 |
+
"target_modules": [
|
| 31 |
+
"q_proj",
|
| 32 |
+
"k_proj",
|
| 33 |
+
"v_proj",
|
| 34 |
+
"o_proj",
|
| 35 |
+
"gate_proj",
|
| 36 |
+
"up_proj",
|
| 37 |
+
"down_proj"
|
| 38 |
+
],
|
| 39 |
+
"warmup_ratio": 0.03,
|
| 40 |
+
"weight_decay": 0.0
|
| 41 |
+
}
|
metadata/mistral7b_lora_alpaca/metrics.json
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_file": null,
|
| 3 |
+
"dataset_name": "tatsu-lab/alpaca",
|
| 4 |
+
"effective_batch_size": 16,
|
| 5 |
+
"eval_metrics": {
|
| 6 |
+
"epoch": 4.0,
|
| 7 |
+
"eval_loss": 1.6778709888458252,
|
| 8 |
+
"eval_runtime": 9.7536,
|
| 9 |
+
"eval_samples_per_second": 26.247,
|
| 10 |
+
"eval_steps_per_second": 13.123,
|
| 11 |
+
"perplexity": 5.354144794182875
|
| 12 |
+
},
|
| 13 |
+
"learning_rate": 0.0002,
|
| 14 |
+
"max_eval_samples": 256,
|
| 15 |
+
"max_seq_length": 512,
|
| 16 |
+
"max_steps": 500,
|
| 17 |
+
"max_train_samples": 2000,
|
| 18 |
+
"method": "lora",
|
| 19 |
+
"model_name": "mistralai/Mistral-7B-v0.1",
|
| 20 |
+
"parameter_counts": {
|
| 21 |
+
"total_params": 7283675136,
|
| 22 |
+
"trainable_params": 41943040,
|
| 23 |
+
"trainable_percent": 0.5758499550960753
|
| 24 |
+
},
|
| 25 |
+
"peak_gpu_memory_gb": 14.497857093811035,
|
| 26 |
+
"seed": 42,
|
| 27 |
+
"train_metrics": {
|
| 28 |
+
"epoch": 4.0,
|
| 29 |
+
"total_flos": 4.216442382424474e+16,
|
| 30 |
+
"train_loss": 0.24137365317344667,
|
| 31 |
+
"train_runtime": 2464.6935,
|
| 32 |
+
"train_samples_per_second": 3.246,
|
| 33 |
+
"train_steps_per_second": 0.203
|
| 34 |
+
},
|
| 35 |
+
"wall_clock_seconds": 2465.9031274318695
|
| 36 |
+
}
|
metadata/mistral7b_lora_alpaca_cot/args.json
ADDED
|
@@ -0,0 +1,41 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"bf16": true,
|
| 3 |
+
"config": "configs/mistral7b_lora_alpaca_cot.yaml",
|
| 4 |
+
"dataset_file": "combination/alcapa_plus_cot.json",
|
| 5 |
+
"dataset_name": "QingyiSi/Alpaca-CoT",
|
| 6 |
+
"dataset_split": "train",
|
| 7 |
+
"eval_steps": 50,
|
| 8 |
+
"gradient_accumulation_steps": 16,
|
| 9 |
+
"gradient_checkpointing": true,
|
| 10 |
+
"learning_rate": 0.0002,
|
| 11 |
+
"logging_steps": 10,
|
| 12 |
+
"lora_alpha": 32,
|
| 13 |
+
"lora_dropout": 0.05,
|
| 14 |
+
"lora_r": 16,
|
| 15 |
+
"max_eval_samples": 256,
|
| 16 |
+
"max_seq_length": 512,
|
| 17 |
+
"max_steps": 500,
|
| 18 |
+
"max_train_samples": 2000,
|
| 19 |
+
"method": "lora",
|
| 20 |
+
"model_name": "mistralai/Mistral-7B-v0.1",
|
| 21 |
+
"num_train_epochs": 1,
|
| 22 |
+
"num_virtual_tokens": 32,
|
| 23 |
+
"output_dir": "outputs/mistral7b_lora_alpaca_cot",
|
| 24 |
+
"per_device_eval_batch_size": 2,
|
| 25 |
+
"per_device_train_batch_size": 1,
|
| 26 |
+
"prompt_tuning_init_text": "Follow the instruction carefully and provide a helpful, accurate response.",
|
| 27 |
+
"resume_from_checkpoint": null,
|
| 28 |
+
"save_steps": 250,
|
| 29 |
+
"seed": 42,
|
| 30 |
+
"target_modules": [
|
| 31 |
+
"q_proj",
|
| 32 |
+
"k_proj",
|
| 33 |
+
"v_proj",
|
| 34 |
+
"o_proj",
|
| 35 |
+
"gate_proj",
|
| 36 |
+
"up_proj",
|
| 37 |
+
"down_proj"
|
| 38 |
+
],
|
| 39 |
+
"warmup_ratio": 0.03,
|
| 40 |
+
"weight_decay": 0.0
|
| 41 |
+
}
|
metadata/mistral7b_lora_alpaca_cot/metrics.json
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_file": "combination/alcapa_plus_cot.json",
|
| 3 |
+
"dataset_name": "QingyiSi/Alpaca-CoT",
|
| 4 |
+
"effective_batch_size": 16,
|
| 5 |
+
"eval_metrics": {
|
| 6 |
+
"epoch": 4.0,
|
| 7 |
+
"eval_loss": 1.5489877462387085,
|
| 8 |
+
"eval_runtime": 10.5771,
|
| 9 |
+
"eval_samples_per_second": 24.109,
|
| 10 |
+
"eval_steps_per_second": 12.102,
|
| 11 |
+
"perplexity": 4.706703392184987
|
| 12 |
+
},
|
| 13 |
+
"learning_rate": 0.0002,
|
| 14 |
+
"max_eval_samples": 256,
|
| 15 |
+
"max_seq_length": 512,
|
| 16 |
+
"max_steps": 500,
|
| 17 |
+
"max_train_samples": 2000,
|
| 18 |
+
"method": "lora",
|
| 19 |
+
"model_name": "mistralai/Mistral-7B-v0.1",
|
| 20 |
+
"parameter_counts": {
|
| 21 |
+
"total_params": 7283675136,
|
| 22 |
+
"trainable_params": 41943040,
|
| 23 |
+
"trainable_percent": 0.5758499550960753
|
| 24 |
+
},
|
| 25 |
+
"peak_gpu_memory_gb": 14.497621536254883,
|
| 26 |
+
"seed": 42,
|
| 27 |
+
"train_metrics": {
|
| 28 |
+
"epoch": 4.0,
|
| 29 |
+
"total_flos": 4.803390717847142e+16,
|
| 30 |
+
"train_loss": 0.4384129907488823,
|
| 31 |
+
"train_runtime": 3637.731,
|
| 32 |
+
"train_samples_per_second": 2.199,
|
| 33 |
+
"train_steps_per_second": 0.137
|
| 34 |
+
},
|
| 35 |
+
"wall_clock_seconds": 3637.9769699573517
|
| 36 |
+
}
|
metadata/mistral7b_lora_guanaco/args.json
ADDED
|
@@ -0,0 +1,41 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"bf16": true,
|
| 3 |
+
"config": "configs/mistral7b_lora_guanaco.yaml",
|
| 4 |
+
"dataset_file": null,
|
| 5 |
+
"dataset_name": "fengtc/GuanacoDataset",
|
| 6 |
+
"dataset_split": "train",
|
| 7 |
+
"eval_steps": 50,
|
| 8 |
+
"gradient_accumulation_steps": 16,
|
| 9 |
+
"gradient_checkpointing": true,
|
| 10 |
+
"learning_rate": 0.0002,
|
| 11 |
+
"logging_steps": 10,
|
| 12 |
+
"lora_alpha": 32,
|
| 13 |
+
"lora_dropout": 0.05,
|
| 14 |
+
"lora_r": 16,
|
| 15 |
+
"max_eval_samples": 256,
|
| 16 |
+
"max_seq_length": 512,
|
| 17 |
+
"max_steps": 500,
|
| 18 |
+
"max_train_samples": 2000,
|
| 19 |
+
"method": "lora",
|
| 20 |
+
"model_name": "mistralai/Mistral-7B-v0.1",
|
| 21 |
+
"num_train_epochs": 1,
|
| 22 |
+
"num_virtual_tokens": 32,
|
| 23 |
+
"output_dir": "outputs/mistral7b_lora_guanaco",
|
| 24 |
+
"per_device_eval_batch_size": 2,
|
| 25 |
+
"per_device_train_batch_size": 1,
|
| 26 |
+
"prompt_tuning_init_text": "Follow the instruction carefully and provide a helpful, accurate response.",
|
| 27 |
+
"resume_from_checkpoint": null,
|
| 28 |
+
"save_steps": 250,
|
| 29 |
+
"seed": 42,
|
| 30 |
+
"target_modules": [
|
| 31 |
+
"q_proj",
|
| 32 |
+
"k_proj",
|
| 33 |
+
"v_proj",
|
| 34 |
+
"o_proj",
|
| 35 |
+
"gate_proj",
|
| 36 |
+
"up_proj",
|
| 37 |
+
"down_proj"
|
| 38 |
+
],
|
| 39 |
+
"warmup_ratio": 0.03,
|
| 40 |
+
"weight_decay": 0.0
|
| 41 |
+
}
|
metadata/mistral7b_lora_guanaco/metrics.json
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_file": null,
|
| 3 |
+
"dataset_name": "fengtc/GuanacoDataset",
|
| 4 |
+
"effective_batch_size": 16,
|
| 5 |
+
"eval_metrics": {
|
| 6 |
+
"epoch": 4.0,
|
| 7 |
+
"eval_loss": 1.4493114948272705,
|
| 8 |
+
"eval_runtime": 13.4923,
|
| 9 |
+
"eval_samples_per_second": 18.974,
|
| 10 |
+
"eval_steps_per_second": 9.487,
|
| 11 |
+
"perplexity": 4.2601803489833925
|
| 12 |
+
},
|
| 13 |
+
"learning_rate": 0.0002,
|
| 14 |
+
"max_eval_samples": 256,
|
| 15 |
+
"max_seq_length": 512,
|
| 16 |
+
"max_steps": 500,
|
| 17 |
+
"max_train_samples": 2000,
|
| 18 |
+
"method": "lora",
|
| 19 |
+
"model_name": "mistralai/Mistral-7B-v0.1",
|
| 20 |
+
"parameter_counts": {
|
| 21 |
+
"total_params": 7283675136,
|
| 22 |
+
"trainable_params": 41943040,
|
| 23 |
+
"trainable_percent": 0.5758499550960753
|
| 24 |
+
},
|
| 25 |
+
"peak_gpu_memory_gb": 14.498394012451172,
|
| 26 |
+
"seed": 42,
|
| 27 |
+
"train_metrics": {
|
| 28 |
+
"epoch": 4.0,
|
| 29 |
+
"total_flos": 7.071807330975744e+16,
|
| 30 |
+
"train_loss": 0.5458194477558136,
|
| 31 |
+
"train_runtime": 3848.5515,
|
| 32 |
+
"train_samples_per_second": 2.079,
|
| 33 |
+
"train_steps_per_second": 0.13
|
| 34 |
+
},
|
| 35 |
+
"wall_clock_seconds": 3848.8962047100067
|
| 36 |
+
}
|
metadata/mistral7b_prompt_alpaca_cot/args.json
ADDED
|
@@ -0,0 +1,33 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"bf16": true,
|
| 3 |
+
"config": "configs/mistral7b_prompt_alpaca_cot.yaml",
|
| 4 |
+
"dataset_file": "combination/alcapa_plus_cot.json",
|
| 5 |
+
"dataset_name": "QingyiSi/Alpaca-CoT",
|
| 6 |
+
"dataset_split": "train",
|
| 7 |
+
"eval_steps": 50,
|
| 8 |
+
"gradient_accumulation_steps": 16,
|
| 9 |
+
"gradient_checkpointing": true,
|
| 10 |
+
"learning_rate": 0.0002,
|
| 11 |
+
"logging_steps": 10,
|
| 12 |
+
"lora_alpha": 32,
|
| 13 |
+
"lora_dropout": 0.05,
|
| 14 |
+
"lora_r": 16,
|
| 15 |
+
"max_eval_samples": 256,
|
| 16 |
+
"max_seq_length": 512,
|
| 17 |
+
"max_steps": 500,
|
| 18 |
+
"max_train_samples": 2000,
|
| 19 |
+
"method": "prompt_tuning",
|
| 20 |
+
"model_name": "mistralai/Mistral-7B-v0.1",
|
| 21 |
+
"num_train_epochs": 1,
|
| 22 |
+
"num_virtual_tokens": 32,
|
| 23 |
+
"output_dir": "outputs/mistral7b_prompt_alpaca_cot",
|
| 24 |
+
"per_device_eval_batch_size": 2,
|
| 25 |
+
"per_device_train_batch_size": 1,
|
| 26 |
+
"prompt_tuning_init_text": "Follow the instruction carefully and provide a helpful, accurate response.",
|
| 27 |
+
"resume_from_checkpoint": null,
|
| 28 |
+
"save_steps": 250,
|
| 29 |
+
"seed": 42,
|
| 30 |
+
"target_modules": null,
|
| 31 |
+
"warmup_ratio": 0.03,
|
| 32 |
+
"weight_decay": 0.0
|
| 33 |
+
}
|
metadata/mistral7b_prompt_alpaca_cot/metrics.json
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_file": "combination/alcapa_plus_cot.json",
|
| 3 |
+
"dataset_name": "QingyiSi/Alpaca-CoT",
|
| 4 |
+
"effective_batch_size": 16,
|
| 5 |
+
"eval_metrics": {
|
| 6 |
+
"epoch": 4.0,
|
| 7 |
+
"eval_loss": 1.012134313583374,
|
| 8 |
+
"eval_runtime": 8.0602,
|
| 9 |
+
"eval_samples_per_second": 31.637,
|
| 10 |
+
"eval_steps_per_second": 15.88,
|
| 11 |
+
"perplexity": 2.7514672465197147
|
| 12 |
+
},
|
| 13 |
+
"learning_rate": 0.0002,
|
| 14 |
+
"max_eval_samples": 256,
|
| 15 |
+
"max_seq_length": 512,
|
| 16 |
+
"max_steps": 500,
|
| 17 |
+
"max_train_samples": 2000,
|
| 18 |
+
"method": "prompt_tuning",
|
| 19 |
+
"model_name": "mistralai/Mistral-7B-v0.1",
|
| 20 |
+
"parameter_counts": {
|
| 21 |
+
"total_params": 7241863168,
|
| 22 |
+
"trainable_params": 131072,
|
| 23 |
+
"trainable_percent": 0.0018099209686697024
|
| 24 |
+
},
|
| 25 |
+
"peak_gpu_memory_gb": 13.847609996795654,
|
| 26 |
+
"seed": 42,
|
| 27 |
+
"train_metrics": {
|
| 28 |
+
"epoch": 4.0,
|
| 29 |
+
"total_flos": 4.775223517013606e+16,
|
| 30 |
+
"train_loss": 1.0451534767150878,
|
| 31 |
+
"train_runtime": 1522.2622,
|
| 32 |
+
"train_samples_per_second": 5.255,
|
| 33 |
+
"train_steps_per_second": 0.328
|
| 34 |
+
},
|
| 35 |
+
"wall_clock_seconds": 1522.533498287201
|
| 36 |
+
}
|
metadata/mistral7b_prompt_guanaco/args.json
ADDED
|
@@ -0,0 +1,33 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"bf16": true,
|
| 3 |
+
"config": "configs/mistral7b_prompt_guanaco.yaml",
|
| 4 |
+
"dataset_file": null,
|
| 5 |
+
"dataset_name": "fengtc/GuanacoDataset",
|
| 6 |
+
"dataset_split": "train",
|
| 7 |
+
"eval_steps": 50,
|
| 8 |
+
"gradient_accumulation_steps": 16,
|
| 9 |
+
"gradient_checkpointing": true,
|
| 10 |
+
"learning_rate": 0.0002,
|
| 11 |
+
"logging_steps": 10,
|
| 12 |
+
"lora_alpha": 32,
|
| 13 |
+
"lora_dropout": 0.05,
|
| 14 |
+
"lora_r": 16,
|
| 15 |
+
"max_eval_samples": 256,
|
| 16 |
+
"max_seq_length": 512,
|
| 17 |
+
"max_steps": 500,
|
| 18 |
+
"max_train_samples": 2000,
|
| 19 |
+
"method": "prompt_tuning",
|
| 20 |
+
"model_name": "mistralai/Mistral-7B-v0.1",
|
| 21 |
+
"num_train_epochs": 1,
|
| 22 |
+
"num_virtual_tokens": 32,
|
| 23 |
+
"output_dir": "outputs/mistral7b_prompt_guanaco",
|
| 24 |
+
"per_device_eval_batch_size": 2,
|
| 25 |
+
"per_device_train_batch_size": 1,
|
| 26 |
+
"prompt_tuning_init_text": "Follow the instruction carefully and provide a helpful, accurate response.",
|
| 27 |
+
"resume_from_checkpoint": null,
|
| 28 |
+
"save_steps": 250,
|
| 29 |
+
"seed": 42,
|
| 30 |
+
"target_modules": null,
|
| 31 |
+
"warmup_ratio": 0.03,
|
| 32 |
+
"weight_decay": 0.0
|
| 33 |
+
}
|
metadata/mistral7b_prompt_guanaco/metrics.json
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_file": null,
|
| 3 |
+
"dataset_name": "fengtc/GuanacoDataset",
|
| 4 |
+
"effective_batch_size": 16,
|
| 5 |
+
"eval_metrics": {
|
| 6 |
+
"epoch": 4.0,
|
| 7 |
+
"eval_loss": 1.0030479431152344,
|
| 8 |
+
"eval_runtime": 10.7067,
|
| 9 |
+
"eval_samples_per_second": 23.91,
|
| 10 |
+
"eval_steps_per_second": 11.955,
|
| 11 |
+
"perplexity": 2.7265796360422554
|
| 12 |
+
},
|
| 13 |
+
"learning_rate": 0.0002,
|
| 14 |
+
"max_eval_samples": 256,
|
| 15 |
+
"max_seq_length": 512,
|
| 16 |
+
"max_steps": 500,
|
| 17 |
+
"max_train_samples": 2000,
|
| 18 |
+
"method": "prompt_tuning",
|
| 19 |
+
"model_name": "mistralai/Mistral-7B-v0.1",
|
| 20 |
+
"parameter_counts": {
|
| 21 |
+
"total_params": 7241863168,
|
| 22 |
+
"trainable_params": 131072,
|
| 23 |
+
"trainable_percent": 0.0018099209686697024
|
| 24 |
+
},
|
| 25 |
+
"peak_gpu_memory_gb": 13.847609519958496,
|
| 26 |
+
"seed": 42,
|
| 27 |
+
"train_metrics": {
|
| 28 |
+
"epoch": 4.0,
|
| 29 |
+
"total_flos": 7.030338079555584e+16,
|
| 30 |
+
"train_loss": 1.0741594829559327,
|
| 31 |
+
"train_runtime": 1610.509,
|
| 32 |
+
"train_samples_per_second": 4.967,
|
| 33 |
+
"train_steps_per_second": 0.31
|
| 34 |
+
},
|
| 35 |
+
"wall_clock_seconds": 1610.6813061237335
|
| 36 |
+
}
|
metadata/mistral7b_prompt_tuning_alpaca/args.json
ADDED
|
@@ -0,0 +1,33 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"bf16": true,
|
| 3 |
+
"config": "configs/mistral_prompt_tuning_alpaca.yaml",
|
| 4 |
+
"dataset_file": null,
|
| 5 |
+
"dataset_name": "tatsu-lab/alpaca",
|
| 6 |
+
"dataset_split": "train",
|
| 7 |
+
"eval_steps": 50,
|
| 8 |
+
"gradient_accumulation_steps": 16,
|
| 9 |
+
"gradient_checkpointing": true,
|
| 10 |
+
"learning_rate": 0.0002,
|
| 11 |
+
"logging_steps": 10,
|
| 12 |
+
"lora_alpha": 32,
|
| 13 |
+
"lora_dropout": 0.05,
|
| 14 |
+
"lora_r": 16,
|
| 15 |
+
"max_eval_samples": 256,
|
| 16 |
+
"max_seq_length": 512,
|
| 17 |
+
"max_steps": 500,
|
| 18 |
+
"max_train_samples": 2000,
|
| 19 |
+
"method": "prompt_tuning",
|
| 20 |
+
"model_name": "mistralai/Mistral-7B-v0.1",
|
| 21 |
+
"num_train_epochs": 1,
|
| 22 |
+
"num_virtual_tokens": 32,
|
| 23 |
+
"output_dir": "outputs/mistral7b_prompt_tuning_alpaca",
|
| 24 |
+
"per_device_eval_batch_size": 2,
|
| 25 |
+
"per_device_train_batch_size": 1,
|
| 26 |
+
"prompt_tuning_init_text": "Follow the instruction carefully and provide a helpful, accurate response.",
|
| 27 |
+
"resume_from_checkpoint": "outputs/mistral7b_prompt_tuning_alpaca/checkpoint-250",
|
| 28 |
+
"save_steps": 250,
|
| 29 |
+
"seed": 42,
|
| 30 |
+
"target_modules": null,
|
| 31 |
+
"warmup_ratio": 0.03,
|
| 32 |
+
"weight_decay": 0.0
|
| 33 |
+
}
|
metadata/mistral7b_prompt_tuning_alpaca/metrics.json
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_file": null,
|
| 3 |
+
"dataset_name": "tatsu-lab/alpaca",
|
| 4 |
+
"effective_batch_size": 16,
|
| 5 |
+
"eval_metrics": {
|
| 6 |
+
"epoch": 4.0,
|
| 7 |
+
"eval_loss": 1.0023709535598755,
|
| 8 |
+
"eval_runtime": 6.1148,
|
| 9 |
+
"eval_samples_per_second": 41.866,
|
| 10 |
+
"eval_steps_per_second": 20.933,
|
| 11 |
+
"perplexity": 2.724734394781806
|
| 12 |
+
},
|
| 13 |
+
"learning_rate": 0.0002,
|
| 14 |
+
"max_eval_samples": 256,
|
| 15 |
+
"max_seq_length": 512,
|
| 16 |
+
"max_steps": 500,
|
| 17 |
+
"max_train_samples": 2000,
|
| 18 |
+
"method": "prompt_tuning",
|
| 19 |
+
"model_name": "mistralai/Mistral-7B-v0.1",
|
| 20 |
+
"parameter_counts": {
|
| 21 |
+
"total_params": 7241863168,
|
| 22 |
+
"trainable_params": 131072,
|
| 23 |
+
"trainable_percent": 0.0018099209686697024
|
| 24 |
+
},
|
| 25 |
+
"peak_gpu_memory_gb": 13.848701477050781,
|
| 26 |
+
"seed": 42,
|
| 27 |
+
"train_metrics": {
|
| 28 |
+
"epoch": 4.0,
|
| 29 |
+
"total_flos": 4.19171706100777e+16,
|
| 30 |
+
"train_loss": 0.5351699771881103,
|
| 31 |
+
"train_runtime": 725.0655,
|
| 32 |
+
"train_samples_per_second": 11.033,
|
| 33 |
+
"train_steps_per_second": 0.69
|
| 34 |
+
},
|
| 35 |
+
"wall_clock_seconds": 725.2680373191833
|
| 36 |
+
}
|
metadata/mistral7b_qlora_alpaca/args.json
ADDED
|
@@ -0,0 +1,41 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"bf16": true,
|
| 3 |
+
"config": "configs/mistral_qlora_alpaca.yaml",
|
| 4 |
+
"dataset_file": null,
|
| 5 |
+
"dataset_name": "tatsu-lab/alpaca",
|
| 6 |
+
"dataset_split": "train",
|
| 7 |
+
"eval_steps": 50,
|
| 8 |
+
"gradient_accumulation_steps": 16,
|
| 9 |
+
"gradient_checkpointing": true,
|
| 10 |
+
"learning_rate": 0.0002,
|
| 11 |
+
"logging_steps": 10,
|
| 12 |
+
"lora_alpha": 32,
|
| 13 |
+
"lora_dropout": 0.05,
|
| 14 |
+
"lora_r": 16,
|
| 15 |
+
"max_eval_samples": 256,
|
| 16 |
+
"max_seq_length": 512,
|
| 17 |
+
"max_steps": 500,
|
| 18 |
+
"max_train_samples": 2000,
|
| 19 |
+
"method": "qlora",
|
| 20 |
+
"model_name": "mistralai/Mistral-7B-v0.1",
|
| 21 |
+
"num_train_epochs": 1,
|
| 22 |
+
"num_virtual_tokens": 32,
|
| 23 |
+
"output_dir": "outputs/mistral7b_qlora_alpaca",
|
| 24 |
+
"per_device_eval_batch_size": 2,
|
| 25 |
+
"per_device_train_batch_size": 1,
|
| 26 |
+
"prompt_tuning_init_text": "Follow the instruction carefully and provide a helpful, accurate response.",
|
| 27 |
+
"resume_from_checkpoint": null,
|
| 28 |
+
"save_steps": 250,
|
| 29 |
+
"seed": 42,
|
| 30 |
+
"target_modules": [
|
| 31 |
+
"q_proj",
|
| 32 |
+
"k_proj",
|
| 33 |
+
"v_proj",
|
| 34 |
+
"o_proj",
|
| 35 |
+
"gate_proj",
|
| 36 |
+
"up_proj",
|
| 37 |
+
"down_proj"
|
| 38 |
+
],
|
| 39 |
+
"warmup_ratio": 0.03,
|
| 40 |
+
"weight_decay": 0.0
|
| 41 |
+
}
|
metadata/mistral7b_qlora_alpaca/metrics.json
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_file": null,
|
| 3 |
+
"dataset_name": "tatsu-lab/alpaca",
|
| 4 |
+
"effective_batch_size": 16,
|
| 5 |
+
"eval_metrics": {
|
| 6 |
+
"epoch": 4.0,
|
| 7 |
+
"eval_loss": 1.743696689605713,
|
| 8 |
+
"eval_runtime": 15.1588,
|
| 9 |
+
"eval_samples_per_second": 16.888,
|
| 10 |
+
"eval_steps_per_second": 8.444,
|
| 11 |
+
"perplexity": 5.718443709459332
|
| 12 |
+
},
|
| 13 |
+
"learning_rate": 0.0002,
|
| 14 |
+
"max_eval_samples": 256,
|
| 15 |
+
"max_seq_length": 512,
|
| 16 |
+
"max_steps": 500,
|
| 17 |
+
"max_train_samples": 2000,
|
| 18 |
+
"method": "qlora",
|
| 19 |
+
"model_name": "mistralai/Mistral-7B-v0.1",
|
| 20 |
+
"parameter_counts": {
|
| 21 |
+
"total_params": 3794014208,
|
| 22 |
+
"trainable_params": 41943040,
|
| 23 |
+
"trainable_percent": 1.1055056122762943
|
| 24 |
+
},
|
| 25 |
+
"peak_gpu_memory_gb": 5.718385696411133,
|
| 26 |
+
"seed": 42,
|
| 27 |
+
"train_metrics": {
|
| 28 |
+
"epoch": 4.0,
|
| 29 |
+
"total_flos": 4.216442382424474e+16,
|
| 30 |
+
"train_loss": 0.5023761793375016,
|
| 31 |
+
"train_runtime": 4203.0874,
|
| 32 |
+
"train_samples_per_second": 1.903,
|
| 33 |
+
"train_steps_per_second": 0.119
|
| 34 |
+
},
|
| 35 |
+
"wall_clock_seconds": 4203.367736339569
|
| 36 |
+
}
|
metadata/mistral7b_qlora_alpaca_cot/args.json
ADDED
|
@@ -0,0 +1,41 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"bf16": true,
|
| 3 |
+
"config": "configs/mistral7b_qlora_alpaca_cot.yaml",
|
| 4 |
+
"dataset_file": "combination/alcapa_plus_cot.json",
|
| 5 |
+
"dataset_name": "QingyiSi/Alpaca-CoT",
|
| 6 |
+
"dataset_split": "train",
|
| 7 |
+
"eval_steps": 50,
|
| 8 |
+
"gradient_accumulation_steps": 16,
|
| 9 |
+
"gradient_checkpointing": true,
|
| 10 |
+
"learning_rate": 0.0002,
|
| 11 |
+
"logging_steps": 10,
|
| 12 |
+
"lora_alpha": 32,
|
| 13 |
+
"lora_dropout": 0.05,
|
| 14 |
+
"lora_r": 16,
|
| 15 |
+
"max_eval_samples": 256,
|
| 16 |
+
"max_seq_length": 512,
|
| 17 |
+
"max_steps": 500,
|
| 18 |
+
"max_train_samples": 2000,
|
| 19 |
+
"method": "qlora",
|
| 20 |
+
"model_name": "mistralai/Mistral-7B-v0.1",
|
| 21 |
+
"num_train_epochs": 1,
|
| 22 |
+
"num_virtual_tokens": 32,
|
| 23 |
+
"output_dir": "outputs/mistral7b_qlora_alpaca_cot",
|
| 24 |
+
"per_device_eval_batch_size": 2,
|
| 25 |
+
"per_device_train_batch_size": 1,
|
| 26 |
+
"prompt_tuning_init_text": "Follow the instruction carefully and provide a helpful, accurate response.",
|
| 27 |
+
"resume_from_checkpoint": null,
|
| 28 |
+
"save_steps": 250,
|
| 29 |
+
"seed": 42,
|
| 30 |
+
"target_modules": [
|
| 31 |
+
"q_proj",
|
| 32 |
+
"k_proj",
|
| 33 |
+
"v_proj",
|
| 34 |
+
"o_proj",
|
| 35 |
+
"gate_proj",
|
| 36 |
+
"up_proj",
|
| 37 |
+
"down_proj"
|
| 38 |
+
],
|
| 39 |
+
"warmup_ratio": 0.03,
|
| 40 |
+
"weight_decay": 0.0
|
| 41 |
+
}
|
metadata/mistral7b_qlora_alpaca_cot/metrics.json
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_file": "combination/alcapa_plus_cot.json",
|
| 3 |
+
"dataset_name": "QingyiSi/Alpaca-CoT",
|
| 4 |
+
"effective_batch_size": 16,
|
| 5 |
+
"eval_metrics": {
|
| 6 |
+
"epoch": 4.0,
|
| 7 |
+
"eval_loss": 1.5378942489624023,
|
| 8 |
+
"eval_runtime": 14.9658,
|
| 9 |
+
"eval_samples_per_second": 17.039,
|
| 10 |
+
"eval_steps_per_second": 8.553,
|
| 11 |
+
"perplexity": 4.654778139763514
|
| 12 |
+
},
|
| 13 |
+
"learning_rate": 0.0002,
|
| 14 |
+
"max_eval_samples": 256,
|
| 15 |
+
"max_seq_length": 512,
|
| 16 |
+
"max_steps": 500,
|
| 17 |
+
"max_train_samples": 2000,
|
| 18 |
+
"method": "qlora",
|
| 19 |
+
"model_name": "mistralai/Mistral-7B-v0.1",
|
| 20 |
+
"parameter_counts": {
|
| 21 |
+
"total_params": 3794014208,
|
| 22 |
+
"trainable_params": 41943040,
|
| 23 |
+
"trainable_percent": 1.1055056122762943
|
| 24 |
+
},
|
| 25 |
+
"peak_gpu_memory_gb": 5.718398094177246,
|
| 26 |
+
"seed": 42,
|
| 27 |
+
"train_metrics": {
|
| 28 |
+
"epoch": 4.0,
|
| 29 |
+
"total_flos": 4.803390717847142e+16,
|
| 30 |
+
"train_loss": 0.4474883066415787,
|
| 31 |
+
"train_runtime": 4813.4527,
|
| 32 |
+
"train_samples_per_second": 1.662,
|
| 33 |
+
"train_steps_per_second": 0.104
|
| 34 |
+
},
|
| 35 |
+
"wall_clock_seconds": 4813.7332644462585
|
| 36 |
+
}
|
metadata/mistral7b_qlora_guanaco/args.json
ADDED
|
@@ -0,0 +1,41 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"bf16": true,
|
| 3 |
+
"config": "configs/mistral7b_qlora_guanaco.yaml",
|
| 4 |
+
"dataset_file": null,
|
| 5 |
+
"dataset_name": "fengtc/GuanacoDataset",
|
| 6 |
+
"dataset_split": "train",
|
| 7 |
+
"eval_steps": 50,
|
| 8 |
+
"gradient_accumulation_steps": 16,
|
| 9 |
+
"gradient_checkpointing": true,
|
| 10 |
+
"learning_rate": 0.0002,
|
| 11 |
+
"logging_steps": 10,
|
| 12 |
+
"lora_alpha": 32,
|
| 13 |
+
"lora_dropout": 0.05,
|
| 14 |
+
"lora_r": 16,
|
| 15 |
+
"max_eval_samples": 256,
|
| 16 |
+
"max_seq_length": 512,
|
| 17 |
+
"max_steps": 500,
|
| 18 |
+
"max_train_samples": 2000,
|
| 19 |
+
"method": "qlora",
|
| 20 |
+
"model_name": "mistralai/Mistral-7B-v0.1",
|
| 21 |
+
"num_train_epochs": 1,
|
| 22 |
+
"num_virtual_tokens": 32,
|
| 23 |
+
"output_dir": "outputs/mistral7b_qlora_guanaco",
|
| 24 |
+
"per_device_eval_batch_size": 2,
|
| 25 |
+
"per_device_train_batch_size": 1,
|
| 26 |
+
"prompt_tuning_init_text": "Follow the instruction carefully and provide a helpful, accurate response.",
|
| 27 |
+
"resume_from_checkpoint": null,
|
| 28 |
+
"save_steps": 250,
|
| 29 |
+
"seed": 42,
|
| 30 |
+
"target_modules": [
|
| 31 |
+
"q_proj",
|
| 32 |
+
"k_proj",
|
| 33 |
+
"v_proj",
|
| 34 |
+
"o_proj",
|
| 35 |
+
"gate_proj",
|
| 36 |
+
"up_proj",
|
| 37 |
+
"down_proj"
|
| 38 |
+
],
|
| 39 |
+
"warmup_ratio": 0.03,
|
| 40 |
+
"weight_decay": 0.0
|
| 41 |
+
}
|
metadata/mistral7b_qlora_guanaco/metrics.json
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_file": null,
|
| 3 |
+
"dataset_name": "fengtc/GuanacoDataset",
|
| 4 |
+
"effective_batch_size": 16,
|
| 5 |
+
"eval_metrics": {
|
| 6 |
+
"epoch": 4.0,
|
| 7 |
+
"eval_loss": 1.4490643739700317,
|
| 8 |
+
"eval_runtime": 16.1985,
|
| 9 |
+
"eval_samples_per_second": 15.804,
|
| 10 |
+
"eval_steps_per_second": 7.902,
|
| 11 |
+
"perplexity": 4.259127699634722
|
| 12 |
+
},
|
| 13 |
+
"learning_rate": 0.0002,
|
| 14 |
+
"max_eval_samples": 256,
|
| 15 |
+
"max_seq_length": 512,
|
| 16 |
+
"max_steps": 500,
|
| 17 |
+
"max_train_samples": 2000,
|
| 18 |
+
"method": "qlora",
|
| 19 |
+
"model_name": "mistralai/Mistral-7B-v0.1",
|
| 20 |
+
"parameter_counts": {
|
| 21 |
+
"total_params": 3794014208,
|
| 22 |
+
"trainable_params": 41943040,
|
| 23 |
+
"trainable_percent": 1.1055056122762943
|
| 24 |
+
},
|
| 25 |
+
"peak_gpu_memory_gb": 5.7177019119262695,
|
| 26 |
+
"seed": 42,
|
| 27 |
+
"train_metrics": {
|
| 28 |
+
"epoch": 4.0,
|
| 29 |
+
"total_flos": 7.071807330975744e+16,
|
| 30 |
+
"train_loss": 0.5492446699142456,
|
| 31 |
+
"train_runtime": 4817.4079,
|
| 32 |
+
"train_samples_per_second": 1.661,
|
| 33 |
+
"train_steps_per_second": 0.104
|
| 34 |
+
},
|
| 35 |
+
"wall_clock_seconds": 4817.6931245327
|
| 36 |
+
}
|
metadata/prompt_tuning_starcoder2_alpaca/args.json
ADDED
|
@@ -0,0 +1,33 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"bf16": true,
|
| 3 |
+
"config": "configs/prompt_tuning.yaml",
|
| 4 |
+
"dataset_file": null,
|
| 5 |
+
"dataset_name": "tatsu-lab/alpaca",
|
| 6 |
+
"dataset_split": "train",
|
| 7 |
+
"eval_steps": 50,
|
| 8 |
+
"gradient_accumulation_steps": 16,
|
| 9 |
+
"gradient_checkpointing": true,
|
| 10 |
+
"learning_rate": 0.0002,
|
| 11 |
+
"logging_steps": 10,
|
| 12 |
+
"lora_alpha": 32,
|
| 13 |
+
"lora_dropout": 0.05,
|
| 14 |
+
"lora_r": 16,
|
| 15 |
+
"max_eval_samples": 256,
|
| 16 |
+
"max_seq_length": 512,
|
| 17 |
+
"max_steps": 500,
|
| 18 |
+
"max_train_samples": 2000,
|
| 19 |
+
"method": "prompt_tuning",
|
| 20 |
+
"model_name": "bigcode/starcoder2-3b",
|
| 21 |
+
"num_train_epochs": 1,
|
| 22 |
+
"num_virtual_tokens": 32,
|
| 23 |
+
"output_dir": "outputs/prompt_tuning_starcoder2_alpaca",
|
| 24 |
+
"per_device_eval_batch_size": 2,
|
| 25 |
+
"per_device_train_batch_size": 1,
|
| 26 |
+
"prompt_tuning_init_text": "Follow the instruction carefully and provide a helpful, accurate response.",
|
| 27 |
+
"resume_from_checkpoint": "outputs/prompt_tuning_starcoder2_alpaca/checkpoint-250",
|
| 28 |
+
"save_steps": 250,
|
| 29 |
+
"seed": 42,
|
| 30 |
+
"target_modules": null,
|
| 31 |
+
"warmup_ratio": 0.03,
|
| 32 |
+
"weight_decay": 0.0
|
| 33 |
+
}
|
metadata/prompt_tuning_starcoder2_alpaca/metrics.json
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_file": null,
|
| 3 |
+
"dataset_name": "tatsu-lab/alpaca",
|
| 4 |
+
"effective_batch_size": 16,
|
| 5 |
+
"eval_metrics": {
|
| 6 |
+
"epoch": 4.0,
|
| 7 |
+
"eval_loss": 1.5601716041564941,
|
| 8 |
+
"eval_runtime": 4.7153,
|
| 9 |
+
"eval_samples_per_second": 54.292,
|
| 10 |
+
"eval_steps_per_second": 27.146,
|
| 11 |
+
"perplexity": 4.759637948716392
|
| 12 |
+
},
|
| 13 |
+
"learning_rate": 0.0002,
|
| 14 |
+
"max_eval_samples": 256,
|
| 15 |
+
"max_seq_length": 512,
|
| 16 |
+
"max_steps": 500,
|
| 17 |
+
"max_train_samples": 2000,
|
| 18 |
+
"method": "prompt_tuning",
|
| 19 |
+
"model_name": "bigcode/starcoder2-3b",
|
| 20 |
+
"parameter_counts": {
|
| 21 |
+
"total_params": 3030469632,
|
| 22 |
+
"trainable_params": 98304,
|
| 23 |
+
"trainable_percent": 0.0032438536575970546
|
| 24 |
+
},
|
| 25 |
+
"peak_gpu_memory_gb": 6.164307594299316,
|
| 26 |
+
"seed": 42,
|
| 27 |
+
"train_metrics": {
|
| 28 |
+
"epoch": 4.0,
|
| 29 |
+
"total_flos": 1.6958927991472128e+16,
|
| 30 |
+
"train_loss": 1.4749931907653808,
|
| 31 |
+
"train_runtime": 599.8836,
|
| 32 |
+
"train_samples_per_second": 13.336,
|
| 33 |
+
"train_steps_per_second": 0.833
|
| 34 |
+
},
|
| 35 |
+
"wall_clock_seconds": 600.0733258724213
|
| 36 |
+
}
|
metadata/qlora_starcoder2_alpaca/args.json
ADDED
|
@@ -0,0 +1,40 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"bf16": true,
|
| 3 |
+
"config": "configs/qlora.yaml",
|
| 4 |
+
"dataset_file": null,
|
| 5 |
+
"dataset_name": "tatsu-lab/alpaca",
|
| 6 |
+
"dataset_split": "train",
|
| 7 |
+
"eval_steps": 50,
|
| 8 |
+
"gradient_accumulation_steps": 16,
|
| 9 |
+
"gradient_checkpointing": true,
|
| 10 |
+
"learning_rate": 0.0002,
|
| 11 |
+
"logging_steps": 10,
|
| 12 |
+
"lora_alpha": 32,
|
| 13 |
+
"lora_dropout": 0.05,
|
| 14 |
+
"lora_r": 16,
|
| 15 |
+
"max_eval_samples": 256,
|
| 16 |
+
"max_seq_length": 512,
|
| 17 |
+
"max_steps": 500,
|
| 18 |
+
"max_train_samples": 2000,
|
| 19 |
+
"method": "qlora",
|
| 20 |
+
"model_name": "bigcode/starcoder2-3b",
|
| 21 |
+
"num_train_epochs": 1,
|
| 22 |
+
"num_virtual_tokens": 32,
|
| 23 |
+
"output_dir": "outputs/qlora_starcoder2_alpaca",
|
| 24 |
+
"per_device_eval_batch_size": 2,
|
| 25 |
+
"per_device_train_batch_size": 1,
|
| 26 |
+
"prompt_tuning_init_text": "Follow the instruction carefully and provide a helpful, accurate response.",
|
| 27 |
+
"resume_from_checkpoint": "outputs/qlora_starcoder2_alpaca/checkpoint-250",
|
| 28 |
+
"save_steps": 250,
|
| 29 |
+
"seed": 42,
|
| 30 |
+
"target_modules": [
|
| 31 |
+
"q_proj",
|
| 32 |
+
"k_proj",
|
| 33 |
+
"v_proj",
|
| 34 |
+
"o_proj",
|
| 35 |
+
"c_fc",
|
| 36 |
+
"c_proj"
|
| 37 |
+
],
|
| 38 |
+
"warmup_ratio": 0.03,
|
| 39 |
+
"weight_decay": 0.0
|
| 40 |
+
}
|
metadata/qlora_starcoder2_alpaca/metrics.json
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_file": null,
|
| 3 |
+
"dataset_name": "tatsu-lab/alpaca",
|
| 4 |
+
"effective_batch_size": 16,
|
| 5 |
+
"eval_metrics": {
|
| 6 |
+
"epoch": 4.0,
|
| 7 |
+
"eval_loss": 1.4460986852645874,
|
| 8 |
+
"eval_runtime": 12.7595,
|
| 9 |
+
"eval_samples_per_second": 20.063,
|
| 10 |
+
"eval_steps_per_second": 10.032,
|
| 11 |
+
"perplexity": 4.2465151643917896
|
| 12 |
+
},
|
| 13 |
+
"learning_rate": 0.0002,
|
| 14 |
+
"max_eval_samples": 256,
|
| 15 |
+
"max_seq_length": 512,
|
| 16 |
+
"max_steps": 500,
|
| 17 |
+
"max_train_samples": 2000,
|
| 18 |
+
"method": "qlora",
|
| 19 |
+
"model_name": "bigcode/starcoder2-3b",
|
| 20 |
+
"parameter_counts": {
|
| 21 |
+
"total_params": 1615039488,
|
| 22 |
+
"trainable_params": 23838720,
|
| 23 |
+
"trainable_percent": 1.4760456432877014
|
| 24 |
+
},
|
| 25 |
+
"peak_gpu_memory_gb": 3.079056739807129,
|
| 26 |
+
"seed": 42,
|
| 27 |
+
"train_metrics": {
|
| 28 |
+
"epoch": 4.0,
|
| 29 |
+
"total_flos": 1.7099333093818368e+16,
|
| 30 |
+
"train_loss": 0.7822464866638184,
|
| 31 |
+
"train_runtime": 1784.5326,
|
| 32 |
+
"train_samples_per_second": 4.483,
|
| 33 |
+
"train_steps_per_second": 0.28
|
| 34 |
+
},
|
| 35 |
+
"wall_clock_seconds": 1785.0911331176758
|
| 36 |
+
}
|
metadata/starcoder2_lora_alpaca_cot/args.json
ADDED
|
@@ -0,0 +1,40 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"bf16": true,
|
| 3 |
+
"config": "configs/starcoder2_lora_alpaca_cot.yaml",
|
| 4 |
+
"dataset_file": "combination/alcapa_plus_cot.json",
|
| 5 |
+
"dataset_name": "QingyiSi/Alpaca-CoT",
|
| 6 |
+
"dataset_split": "train",
|
| 7 |
+
"eval_steps": 50,
|
| 8 |
+
"gradient_accumulation_steps": 16,
|
| 9 |
+
"gradient_checkpointing": true,
|
| 10 |
+
"learning_rate": 0.0002,
|
| 11 |
+
"logging_steps": 10,
|
| 12 |
+
"lora_alpha": 32,
|
| 13 |
+
"lora_dropout": 0.05,
|
| 14 |
+
"lora_r": 16,
|
| 15 |
+
"max_eval_samples": 256,
|
| 16 |
+
"max_seq_length": 512,
|
| 17 |
+
"max_steps": 500,
|
| 18 |
+
"max_train_samples": 2000,
|
| 19 |
+
"method": "lora",
|
| 20 |
+
"model_name": "bigcode/starcoder2-3b",
|
| 21 |
+
"num_train_epochs": 1,
|
| 22 |
+
"num_virtual_tokens": 32,
|
| 23 |
+
"output_dir": "outputs/starcoder2_lora_alpaca_cot",
|
| 24 |
+
"per_device_eval_batch_size": 2,
|
| 25 |
+
"per_device_train_batch_size": 1,
|
| 26 |
+
"prompt_tuning_init_text": "Follow the instruction carefully and provide a helpful, accurate response.",
|
| 27 |
+
"resume_from_checkpoint": null,
|
| 28 |
+
"save_steps": 250,
|
| 29 |
+
"seed": 42,
|
| 30 |
+
"target_modules": [
|
| 31 |
+
"q_proj",
|
| 32 |
+
"k_proj",
|
| 33 |
+
"v_proj",
|
| 34 |
+
"o_proj",
|
| 35 |
+
"c_fc",
|
| 36 |
+
"c_proj"
|
| 37 |
+
],
|
| 38 |
+
"warmup_ratio": 0.03,
|
| 39 |
+
"weight_decay": 0.0
|
| 40 |
+
}
|
metadata/starcoder2_lora_alpaca_cot/metrics.json
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_file": "combination/alcapa_plus_cot.json",
|
| 3 |
+
"dataset_name": "QingyiSi/Alpaca-CoT",
|
| 4 |
+
"effective_batch_size": 16,
|
| 5 |
+
"eval_metrics": {
|
| 6 |
+
"epoch": 4.0,
|
| 7 |
+
"eval_loss": 1.2602890729904175,
|
| 8 |
+
"eval_runtime": 7.8256,
|
| 9 |
+
"eval_samples_per_second": 32.586,
|
| 10 |
+
"eval_steps_per_second": 16.357,
|
| 11 |
+
"perplexity": 3.5264407388091508
|
| 12 |
+
},
|
| 13 |
+
"learning_rate": 0.0002,
|
| 14 |
+
"max_eval_samples": 256,
|
| 15 |
+
"max_seq_length": 512,
|
| 16 |
+
"max_steps": 500,
|
| 17 |
+
"max_train_samples": 2000,
|
| 18 |
+
"method": "lora",
|
| 19 |
+
"model_name": "bigcode/starcoder2-3b",
|
| 20 |
+
"parameter_counts": {
|
| 21 |
+
"total_params": 3054210048,
|
| 22 |
+
"trainable_params": 23838720,
|
| 23 |
+
"trainable_percent": 0.7805199912694414
|
| 24 |
+
},
|
| 25 |
+
"peak_gpu_memory_gb": 6.45817756652832,
|
| 26 |
+
"seed": 42,
|
| 27 |
+
"train_metrics": {
|
| 28 |
+
"epoch": 4.0,
|
| 29 |
+
"total_flos": 1.954360795417805e+16,
|
| 30 |
+
"train_loss": 1.4059115772247315,
|
| 31 |
+
"train_runtime": 2861.1711,
|
| 32 |
+
"train_samples_per_second": 2.796,
|
| 33 |
+
"train_steps_per_second": 0.175
|
| 34 |
+
},
|
| 35 |
+
"wall_clock_seconds": 2861.4430129528046
|
| 36 |
+
}
|
metadata/starcoder2_lora_guanaco/args.json
ADDED
|
@@ -0,0 +1,40 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"bf16": true,
|
| 3 |
+
"config": "configs/starcoder2_lora_guanaco.yaml",
|
| 4 |
+
"dataset_file": null,
|
| 5 |
+
"dataset_name": "fengtc/GuanacoDataset",
|
| 6 |
+
"dataset_split": "train",
|
| 7 |
+
"eval_steps": 50,
|
| 8 |
+
"gradient_accumulation_steps": 16,
|
| 9 |
+
"gradient_checkpointing": true,
|
| 10 |
+
"learning_rate": 0.0002,
|
| 11 |
+
"logging_steps": 10,
|
| 12 |
+
"lora_alpha": 32,
|
| 13 |
+
"lora_dropout": 0.05,
|
| 14 |
+
"lora_r": 16,
|
| 15 |
+
"max_eval_samples": 256,
|
| 16 |
+
"max_seq_length": 512,
|
| 17 |
+
"max_steps": 500,
|
| 18 |
+
"max_train_samples": 2000,
|
| 19 |
+
"method": "lora",
|
| 20 |
+
"model_name": "bigcode/starcoder2-3b",
|
| 21 |
+
"num_train_epochs": 1,
|
| 22 |
+
"num_virtual_tokens": 32,
|
| 23 |
+
"output_dir": "outputs/starcoder2_lora_guanaco",
|
| 24 |
+
"per_device_eval_batch_size": 2,
|
| 25 |
+
"per_device_train_batch_size": 1,
|
| 26 |
+
"prompt_tuning_init_text": "Follow the instruction carefully and provide a helpful, accurate response.",
|
| 27 |
+
"resume_from_checkpoint": null,
|
| 28 |
+
"save_steps": 250,
|
| 29 |
+
"seed": 42,
|
| 30 |
+
"target_modules": [
|
| 31 |
+
"q_proj",
|
| 32 |
+
"k_proj",
|
| 33 |
+
"v_proj",
|
| 34 |
+
"o_proj",
|
| 35 |
+
"c_fc",
|
| 36 |
+
"c_proj"
|
| 37 |
+
],
|
| 38 |
+
"warmup_ratio": 0.03,
|
| 39 |
+
"weight_decay": 0.0
|
| 40 |
+
}
|
metadata/starcoder2_lora_guanaco/metrics.json
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_file": null,
|
| 3 |
+
"dataset_name": "fengtc/GuanacoDataset",
|
| 4 |
+
"effective_batch_size": 16,
|
| 5 |
+
"eval_metrics": {
|
| 6 |
+
"epoch": 4.0,
|
| 7 |
+
"eval_loss": 1.2887365818023682,
|
| 8 |
+
"eval_runtime": 8.8385,
|
| 9 |
+
"eval_samples_per_second": 28.964,
|
| 10 |
+
"eval_steps_per_second": 14.482,
|
| 11 |
+
"perplexity": 3.6281997252628484
|
| 12 |
+
},
|
| 13 |
+
"learning_rate": 0.0002,
|
| 14 |
+
"max_eval_samples": 256,
|
| 15 |
+
"max_seq_length": 512,
|
| 16 |
+
"max_steps": 500,
|
| 17 |
+
"max_train_samples": 2000,
|
| 18 |
+
"method": "lora",
|
| 19 |
+
"model_name": "bigcode/starcoder2-3b",
|
| 20 |
+
"parameter_counts": {
|
| 21 |
+
"total_params": 3054210048,
|
| 22 |
+
"trainable_params": 23838720,
|
| 23 |
+
"trainable_percent": 0.7805199912694414
|
| 24 |
+
},
|
| 25 |
+
"peak_gpu_memory_gb": 6.45782995223999,
|
| 26 |
+
"seed": 42,
|
| 27 |
+
"train_metrics": {
|
| 28 |
+
"epoch": 4.0,
|
| 29 |
+
"total_flos": 2.638869239778509e+16,
|
| 30 |
+
"train_loss": 1.6917072792053223,
|
| 31 |
+
"train_runtime": 2982.7018,
|
| 32 |
+
"train_samples_per_second": 2.682,
|
| 33 |
+
"train_steps_per_second": 0.168
|
| 34 |
+
},
|
| 35 |
+
"wall_clock_seconds": 2982.9298701286316
|
| 36 |
+
}
|
metadata/starcoder2_prompt_alpaca_cot/args.json
ADDED
|
@@ -0,0 +1,33 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"bf16": true,
|
| 3 |
+
"config": "configs/starcoder2_prompt_alpaca_cot.yaml",
|
| 4 |
+
"dataset_file": "combination/alcapa_plus_cot.json",
|
| 5 |
+
"dataset_name": "QingyiSi/Alpaca-CoT",
|
| 6 |
+
"dataset_split": "train",
|
| 7 |
+
"eval_steps": 50,
|
| 8 |
+
"gradient_accumulation_steps": 16,
|
| 9 |
+
"gradient_checkpointing": true,
|
| 10 |
+
"learning_rate": 0.0002,
|
| 11 |
+
"logging_steps": 10,
|
| 12 |
+
"lora_alpha": 32,
|
| 13 |
+
"lora_dropout": 0.05,
|
| 14 |
+
"lora_r": 16,
|
| 15 |
+
"max_eval_samples": 256,
|
| 16 |
+
"max_seq_length": 512,
|
| 17 |
+
"max_steps": 500,
|
| 18 |
+
"max_train_samples": 2000,
|
| 19 |
+
"method": "prompt_tuning",
|
| 20 |
+
"model_name": "bigcode/starcoder2-3b",
|
| 21 |
+
"num_train_epochs": 1,
|
| 22 |
+
"num_virtual_tokens": 32,
|
| 23 |
+
"output_dir": "outputs/starcoder2_prompt_alpaca_cot",
|
| 24 |
+
"per_device_eval_batch_size": 2,
|
| 25 |
+
"per_device_train_batch_size": 1,
|
| 26 |
+
"prompt_tuning_init_text": "Follow the instruction carefully and provide a helpful, accurate response.",
|
| 27 |
+
"resume_from_checkpoint": null,
|
| 28 |
+
"save_steps": 250,
|
| 29 |
+
"seed": 42,
|
| 30 |
+
"target_modules": null,
|
| 31 |
+
"warmup_ratio": 0.03,
|
| 32 |
+
"weight_decay": 0.0
|
| 33 |
+
}
|
metadata/starcoder2_prompt_alpaca_cot/metrics.json
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_file": "combination/alcapa_plus_cot.json",
|
| 3 |
+
"dataset_name": "QingyiSi/Alpaca-CoT",
|
| 4 |
+
"effective_batch_size": 16,
|
| 5 |
+
"eval_metrics": {
|
| 6 |
+
"epoch": 4.0,
|
| 7 |
+
"eval_loss": 1.6411982774734497,
|
| 8 |
+
"eval_runtime": 3.7453,
|
| 9 |
+
"eval_samples_per_second": 68.086,
|
| 10 |
+
"eval_steps_per_second": 34.176,
|
| 11 |
+
"perplexity": 5.161350538285551
|
| 12 |
+
},
|
| 13 |
+
"learning_rate": 0.0002,
|
| 14 |
+
"max_eval_samples": 256,
|
| 15 |
+
"max_seq_length": 512,
|
| 16 |
+
"max_steps": 500,
|
| 17 |
+
"max_train_samples": 2000,
|
| 18 |
+
"method": "prompt_tuning",
|
| 19 |
+
"model_name": "bigcode/starcoder2-3b",
|
| 20 |
+
"parameter_counts": {
|
| 21 |
+
"total_params": 3030469632,
|
| 22 |
+
"trainable_params": 98304,
|
| 23 |
+
"trainable_percent": 0.0032438536575970546
|
| 24 |
+
},
|
| 25 |
+
"peak_gpu_memory_gb": 6.1845526695251465,
|
| 26 |
+
"seed": 42,
|
| 27 |
+
"train_metrics": {
|
| 28 |
+
"epoch": 4.0,
|
| 29 |
+
"total_flos": 1.938313255668941e+16,
|
| 30 |
+
"train_loss": 2.839910514831543,
|
| 31 |
+
"train_runtime": 1108.6114,
|
| 32 |
+
"train_samples_per_second": 7.216,
|
| 33 |
+
"train_steps_per_second": 0.451
|
| 34 |
+
},
|
| 35 |
+
"wall_clock_seconds": 1109.3785409927368
|
| 36 |
+
}
|
metadata/starcoder2_prompt_guanaco/args.json
ADDED
|
@@ -0,0 +1,33 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"bf16": true,
|
| 3 |
+
"config": "configs/starcoder2_prompt_guanaco.yaml",
|
| 4 |
+
"dataset_file": null,
|
| 5 |
+
"dataset_name": "fengtc/GuanacoDataset",
|
| 6 |
+
"dataset_split": "train",
|
| 7 |
+
"eval_steps": 50,
|
| 8 |
+
"gradient_accumulation_steps": 16,
|
| 9 |
+
"gradient_checkpointing": true,
|
| 10 |
+
"learning_rate": 0.0002,
|
| 11 |
+
"logging_steps": 10,
|
| 12 |
+
"lora_alpha": 32,
|
| 13 |
+
"lora_dropout": 0.05,
|
| 14 |
+
"lora_r": 16,
|
| 15 |
+
"max_eval_samples": 256,
|
| 16 |
+
"max_seq_length": 512,
|
| 17 |
+
"max_steps": 500,
|
| 18 |
+
"max_train_samples": 2000,
|
| 19 |
+
"method": "prompt_tuning",
|
| 20 |
+
"model_name": "bigcode/starcoder2-3b",
|
| 21 |
+
"num_train_epochs": 1,
|
| 22 |
+
"num_virtual_tokens": 32,
|
| 23 |
+
"output_dir": "outputs/starcoder2_prompt_guanaco",
|
| 24 |
+
"per_device_eval_batch_size": 2,
|
| 25 |
+
"per_device_train_batch_size": 1,
|
| 26 |
+
"prompt_tuning_init_text": "Follow the instruction carefully and provide a helpful, accurate response.",
|
| 27 |
+
"resume_from_checkpoint": null,
|
| 28 |
+
"save_steps": 250,
|
| 29 |
+
"seed": 42,
|
| 30 |
+
"target_modules": null,
|
| 31 |
+
"warmup_ratio": 0.03,
|
| 32 |
+
"weight_decay": 0.0
|
| 33 |
+
}
|
metadata/starcoder2_prompt_guanaco/metrics.json
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_file": null,
|
| 3 |
+
"dataset_name": "fengtc/GuanacoDataset",
|
| 4 |
+
"effective_batch_size": 16,
|
| 5 |
+
"eval_metrics": {
|
| 6 |
+
"epoch": 4.0,
|
| 7 |
+
"eval_loss": 1.3929762840270996,
|
| 8 |
+
"eval_runtime": 4.401,
|
| 9 |
+
"eval_samples_per_second": 58.169,
|
| 10 |
+
"eval_steps_per_second": 29.085,
|
| 11 |
+
"perplexity": 4.026817187039079
|
| 12 |
+
},
|
| 13 |
+
"learning_rate": 0.0002,
|
| 14 |
+
"max_eval_samples": 256,
|
| 15 |
+
"max_seq_length": 512,
|
| 16 |
+
"max_steps": 500,
|
| 17 |
+
"max_train_samples": 2000,
|
| 18 |
+
"method": "prompt_tuning",
|
| 19 |
+
"model_name": "bigcode/starcoder2-3b",
|
| 20 |
+
"parameter_counts": {
|
| 21 |
+
"total_params": 3030469632,
|
| 22 |
+
"trainable_params": 98304,
|
| 23 |
+
"trainable_percent": 0.0032438536575970546
|
| 24 |
+
},
|
| 25 |
+
"peak_gpu_memory_gb": 6.185425758361816,
|
| 26 |
+
"seed": 42,
|
| 27 |
+
"train_metrics": {
|
| 28 |
+
"epoch": 4.0,
|
| 29 |
+
"total_flos": 2.617201101982925e+16,
|
| 30 |
+
"train_loss": 3.4105590629577636,
|
| 31 |
+
"train_runtime": 1179.8258,
|
| 32 |
+
"train_samples_per_second": 6.781,
|
| 33 |
+
"train_steps_per_second": 0.424
|
| 34 |
+
},
|
| 35 |
+
"wall_clock_seconds": 1180.018343448639
|
| 36 |
+
}
|
metadata/starcoder2_qlora_alpaca_cot/args.json
ADDED
|
@@ -0,0 +1,40 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"bf16": true,
|
| 3 |
+
"config": "configs/starcoder2_qlora_alpaca_cot.yaml",
|
| 4 |
+
"dataset_file": "combination/alcapa_plus_cot.json",
|
| 5 |
+
"dataset_name": "QingyiSi/Alpaca-CoT",
|
| 6 |
+
"dataset_split": "train",
|
| 7 |
+
"eval_steps": 50,
|
| 8 |
+
"gradient_accumulation_steps": 16,
|
| 9 |
+
"gradient_checkpointing": true,
|
| 10 |
+
"learning_rate": 0.0002,
|
| 11 |
+
"logging_steps": 10,
|
| 12 |
+
"lora_alpha": 32,
|
| 13 |
+
"lora_dropout": 0.05,
|
| 14 |
+
"lora_r": 16,
|
| 15 |
+
"max_eval_samples": 256,
|
| 16 |
+
"max_seq_length": 512,
|
| 17 |
+
"max_steps": 500,
|
| 18 |
+
"max_train_samples": 2000,
|
| 19 |
+
"method": "qlora",
|
| 20 |
+
"model_name": "bigcode/starcoder2-3b",
|
| 21 |
+
"num_train_epochs": 1,
|
| 22 |
+
"num_virtual_tokens": 32,
|
| 23 |
+
"output_dir": "outputs/starcoder2_qlora_alpaca_cot",
|
| 24 |
+
"per_device_eval_batch_size": 2,
|
| 25 |
+
"per_device_train_batch_size": 1,
|
| 26 |
+
"prompt_tuning_init_text": "Follow the instruction carefully and provide a helpful, accurate response.",
|
| 27 |
+
"resume_from_checkpoint": "outputs/starcoder2_qlora_alpaca_cot/checkpoint-250",
|
| 28 |
+
"save_steps": 250,
|
| 29 |
+
"seed": 42,
|
| 30 |
+
"target_modules": [
|
| 31 |
+
"q_proj",
|
| 32 |
+
"k_proj",
|
| 33 |
+
"v_proj",
|
| 34 |
+
"o_proj",
|
| 35 |
+
"c_fc",
|
| 36 |
+
"c_proj"
|
| 37 |
+
],
|
| 38 |
+
"warmup_ratio": 0.03,
|
| 39 |
+
"weight_decay": 0.0
|
| 40 |
+
}
|
metadata/starcoder2_qlora_alpaca_cot/metrics.json
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_file": "combination/alcapa_plus_cot.json",
|
| 3 |
+
"dataset_name": "QingyiSi/Alpaca-CoT",
|
| 4 |
+
"effective_batch_size": 16,
|
| 5 |
+
"eval_metrics": {
|
| 6 |
+
"epoch": 4.0,
|
| 7 |
+
"eval_loss": 1.2594980001449585,
|
| 8 |
+
"eval_runtime": 12.4767,
|
| 9 |
+
"eval_samples_per_second": 20.438,
|
| 10 |
+
"eval_steps_per_second": 10.259,
|
| 11 |
+
"perplexity": 3.5236521704253443
|
| 12 |
+
},
|
| 13 |
+
"learning_rate": 0.0002,
|
| 14 |
+
"max_eval_samples": 256,
|
| 15 |
+
"max_seq_length": 512,
|
| 16 |
+
"max_steps": 500,
|
| 17 |
+
"max_train_samples": 2000,
|
| 18 |
+
"method": "qlora",
|
| 19 |
+
"model_name": "bigcode/starcoder2-3b",
|
| 20 |
+
"parameter_counts": {
|
| 21 |
+
"total_params": 1615039488,
|
| 22 |
+
"trainable_params": 23838720,
|
| 23 |
+
"trainable_percent": 1.4760456432877014
|
| 24 |
+
},
|
| 25 |
+
"peak_gpu_memory_gb": 3.0788111686706543,
|
| 26 |
+
"seed": 42,
|
| 27 |
+
"train_metrics": {
|
| 28 |
+
"epoch": 4.0,
|
| 29 |
+
"total_flos": 1.954360795417805e+16,
|
| 30 |
+
"train_loss": 0.6505789184570312,
|
| 31 |
+
"train_runtime": 1766.2618,
|
| 32 |
+
"train_samples_per_second": 4.529,
|
| 33 |
+
"train_steps_per_second": 0.283
|
| 34 |
+
},
|
| 35 |
+
"wall_clock_seconds": 1767.02236866951
|
| 36 |
+
}
|
metadata/starcoder2_qlora_guanaco/args.json
ADDED
|
@@ -0,0 +1,40 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"bf16": true,
|
| 3 |
+
"config": "configs/starcoder2_qlora_guanaco.yaml",
|
| 4 |
+
"dataset_file": null,
|
| 5 |
+
"dataset_name": "fengtc/GuanacoDataset",
|
| 6 |
+
"dataset_split": "train",
|
| 7 |
+
"eval_steps": 50,
|
| 8 |
+
"gradient_accumulation_steps": 16,
|
| 9 |
+
"gradient_checkpointing": true,
|
| 10 |
+
"learning_rate": 0.0002,
|
| 11 |
+
"logging_steps": 10,
|
| 12 |
+
"lora_alpha": 32,
|
| 13 |
+
"lora_dropout": 0.05,
|
| 14 |
+
"lora_r": 16,
|
| 15 |
+
"max_eval_samples": 256,
|
| 16 |
+
"max_seq_length": 512,
|
| 17 |
+
"max_steps": 500,
|
| 18 |
+
"max_train_samples": 2000,
|
| 19 |
+
"method": "qlora",
|
| 20 |
+
"model_name": "bigcode/starcoder2-3b",
|
| 21 |
+
"num_train_epochs": 1,
|
| 22 |
+
"num_virtual_tokens": 32,
|
| 23 |
+
"output_dir": "outputs/starcoder2_qlora_guanaco",
|
| 24 |
+
"per_device_eval_batch_size": 2,
|
| 25 |
+
"per_device_train_batch_size": 1,
|
| 26 |
+
"prompt_tuning_init_text": "Follow the instruction carefully and provide a helpful, accurate response.",
|
| 27 |
+
"resume_from_checkpoint": null,
|
| 28 |
+
"save_steps": 250,
|
| 29 |
+
"seed": 42,
|
| 30 |
+
"target_modules": [
|
| 31 |
+
"q_proj",
|
| 32 |
+
"k_proj",
|
| 33 |
+
"v_proj",
|
| 34 |
+
"o_proj",
|
| 35 |
+
"c_fc",
|
| 36 |
+
"c_proj"
|
| 37 |
+
],
|
| 38 |
+
"warmup_ratio": 0.03,
|
| 39 |
+
"weight_decay": 0.0
|
| 40 |
+
}
|
metadata/starcoder2_qlora_guanaco/metrics.json
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"dataset_file": null,
|
| 3 |
+
"dataset_name": "fengtc/GuanacoDataset",
|
| 4 |
+
"effective_batch_size": 16,
|
| 5 |
+
"eval_metrics": {
|
| 6 |
+
"epoch": 4.0,
|
| 7 |
+
"eval_loss": 1.2983253002166748,
|
| 8 |
+
"eval_runtime": 16.2141,
|
| 9 |
+
"eval_samples_per_second": 15.789,
|
| 10 |
+
"eval_steps_per_second": 7.894,
|
| 11 |
+
"perplexity": 3.6631568399040884
|
| 12 |
+
},
|
| 13 |
+
"learning_rate": 0.0002,
|
| 14 |
+
"max_eval_samples": 256,
|
| 15 |
+
"max_seq_length": 512,
|
| 16 |
+
"max_steps": 500,
|
| 17 |
+
"max_train_samples": 2000,
|
| 18 |
+
"method": "qlora",
|
| 19 |
+
"model_name": "bigcode/starcoder2-3b",
|
| 20 |
+
"parameter_counts": {
|
| 21 |
+
"total_params": 1615039488,
|
| 22 |
+
"trainable_params": 23838720,
|
| 23 |
+
"trainable_percent": 1.4760456432877014
|
| 24 |
+
},
|
| 25 |
+
"peak_gpu_memory_gb": 3.079078197479248,
|
| 26 |
+
"seed": 42,
|
| 27 |
+
"train_metrics": {
|
| 28 |
+
"epoch": 4.0,
|
| 29 |
+
"total_flos": 2.638869239778509e+16,
|
| 30 |
+
"train_loss": 1.715220178604126,
|
| 31 |
+
"train_runtime": 4170.5822,
|
| 32 |
+
"train_samples_per_second": 1.918,
|
| 33 |
+
"train_steps_per_second": 0.12
|
| 34 |
+
},
|
| 35 |
+
"wall_clock_seconds": 4170.863238573074
|
| 36 |
+
}
|