initial commit
Browse files- adapter_config.json +18 -0
- adapter_model.bin +3 -0
- traininglogs.txt +18 -0
adapter_config.json
ADDED
@@ -0,0 +1,18 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"base_model_name_or_path": "decapoda-research/llama-7b-hf",
|
3 |
+
"bias": "none",
|
4 |
+
"enable_lora": null,
|
5 |
+
"fan_in_fan_out": false,
|
6 |
+
"inference_mode": true,
|
7 |
+
"lora_alpha": 16,
|
8 |
+
"lora_dropout": 0.05,
|
9 |
+
"merge_weights": false,
|
10 |
+
"modules_to_save": null,
|
11 |
+
"peft_type": "LORA",
|
12 |
+
"r": 8,
|
13 |
+
"target_modules": [
|
14 |
+
"q_proj",
|
15 |
+
"v_proj"
|
16 |
+
],
|
17 |
+
"task_type": "CAUSAL_LM"
|
18 |
+
}
|
adapter_model.bin
ADDED
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:9e727cf45b015a892bb23deb0e0e016963f52d0106d8996430fc2bd1d9638b4c
|
3 |
+
size 16822989
|
traininglogs.txt
ADDED
@@ -0,0 +1,18 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{'loss': 2.4072, 'learning_rate': 5.9999999999999995e-05, 'epoch': 0.17}
|
2 |
+
{'loss': 1.9372, 'learning_rate': 0.00011399999999999999, 'epoch': 0.34}
|
3 |
+
{'loss': 1.4099, 'learning_rate': 0.00017399999999999997, 'epoch': 0.51}
|
4 |
+
{'loss': 1.2822, 'learning_rate': 0.000234, 'epoch': 0.68}
|
5 |
+
{'loss': 1.2445, 'learning_rate': 0.000294, 'epoch': 0.85}
|
6 |
+
{'loss': 1.2077, 'learning_rate': 0.00027848605577689237, 'epoch': 1.02}
|
7 |
+
{'loss': 1.1843, 'learning_rate': 0.0002545816733067729, 'epoch': 1.19}
|
8 |
+
{'loss': 1.18, 'learning_rate': 0.00023067729083665336, 'epoch': 1.37}
|
9 |
+
{'loss': 1.1533, 'learning_rate': 0.00020677290836653385, 'epoch': 1.54}
|
10 |
+
{'loss': 1.1383, 'learning_rate': 0.00018286852589641432, 'epoch': 1.71}
|
11 |
+
{'loss': 1.1389, 'learning_rate': 0.0001589641434262948, 'epoch': 1.88}
|
12 |
+
{'loss': 1.1331, 'learning_rate': 0.00013505976095617528, 'epoch': 2.05}
|
13 |
+
{'loss': 1.1001, 'learning_rate': 0.00011115537848605577, 'epoch': 2.22}
|
14 |
+
{'loss': 1.1082, 'learning_rate': 8.725099601593625e-05, 'epoch': 2.39}
|
15 |
+
{'loss': 1.0986, 'learning_rate': 6.334661354581673e-05, 'epoch': 2.56}
|
16 |
+
{'loss': 1.0954, 'learning_rate': 3.9442231075697205e-05, 'epoch': 2.73}
|
17 |
+
{'loss': 1.1076, 'learning_rate': 1.5537848605577688e-05, 'epoch': 2.9}
|
18 |
+
{'train_runtime': 9669.9054, 'train_samples_per_second': 4.654, 'train_steps_per_second': 0.036, 'train_loss': 1.283483600344753, 'epoch': 3.0}
|