A-C-E commited on Nov 21, 2023

Commit

bc45211

•

1 Parent(s): d7550f5

Upload 29 files

Browse files

Files changed (18) hide show

checkpoint-1000/config.json +3 -72
checkpoint-1000/generation_config.json +2 -1
checkpoint-1000/model.safetensors +1 -1
checkpoint-1000/optimizer.pt +1 -1
checkpoint-1000/rng_state.pth +1 -1
checkpoint-1000/trainer_state.json +29 -29
checkpoint-1000/training_args.bin +1 -1
checkpoint-500/config.json +3 -72
checkpoint-500/generation_config.json +2 -1
checkpoint-500/model.safetensors +1 -1
checkpoint-500/optimizer.pt +1 -1
checkpoint-500/rng_state.pth +1 -1
checkpoint-500/trainer_state.json +15 -15
checkpoint-500/training_args.bin +1 -1
config.json +3 -72
generation_config.json +2 -1
model.safetensors +1 -1
training_args.bin +1 -1

checkpoint-1000/config.json CHANGED Viewed

@@ -1,5 +1,5 @@
 {
-  "_name_or_path": "google/pegasus-large",
   "activation_dropout": 0.1,
   "activation_function": "relu",
   "add_bias_logits": false,
@@ -10,7 +10,6 @@
   "attention_dropout": 0.1,
   "bos_token_id": 0,
   "classif_dropout": 0.0,
-  "classifier_dropout": 0.0,
   "d_model": 1024,
   "decoder_attention_heads": 16,
   "decoder_ffn_dim": 4096,
@@ -24,9 +23,7 @@
   "encoder_layers": 16,
   "eos_token_id": 1,
   "extra_pos_embeddings": 1,
-  "force_bos_token_to_be_generated": false,
   "forced_eos_token_id": 1,
-  "gradient_checkpointing": false,
   "id2label": {
     "0": "LABEL_0",
     "1": "LABEL_1",
@@ -40,8 +37,9 @@
     "LABEL_2": 2
   },
   "length_penalty": 0.8,
-  "max_length": 256,
   "max_position_embeddings": 1024,
   "model_type": "pegasus",
   "normalize_before": true,
   "normalize_embedding": false,
@@ -50,73 +48,6 @@
   "pad_token_id": 0,
   "scale_embedding": true,
   "static_position_embeddings": true,
-  "task_specific_params": {
-    "summarization_aeslc": {
-      "length_penalty": 0.6,
-      "max_length": 32,
-      "max_position_embeddings": 512
-    },
-    "summarization_arxiv": {
-      "length_penalty": 0.8,
-      "max_length": 256,
-      "max_position_embeddings": 1024
-    },
-    "summarization_big_patent": {
-      "length_penalty": 0.7,
-      "max_length": 256,
-      "max_position_embeddings": 1024
-    },
-    "summarization_billsum": {
-      "length_penalty": 0.6,
-      "max_length": 256,
-      "max_position_embeddings": 1024
-    },
-    "summarization_cnn_dailymail": {
-      "length_penalty": 0.8,
-      "max_length": 128,
-      "max_position_embeddings": 1024
-    },
-    "summarization_gigaword": {
-      "length_penalty": 0.6,
-      "max_length": 32,
-      "max_position_embeddings": 128
-    },
-    "summarization_large": {
-      "length_penalty": 0.8,
-      "max_length": 256,
-      "max_position_embeddings": 1024
-    },
-    "summarization_multi_news": {
-      "length_penalty": 0.8,
-      "max_length": 256,
-      "max_position_embeddings": 1024
-    },
-    "summarization_newsroom": {
-      "length_penalty": 0.8,
-      "max_length": 128,
-      "max_position_embeddings": 512
-    },
-    "summarization_pubmed": {
-      "length_penalty": 0.8,
-      "max_length": 256,
-      "max_position_embeddings": 1024
-    },
-    "summarization_reddit_tifu": {
-      "length_penalty": 0.6,
-      "max_length": 128,
-      "max_position_embeddings": 512
-    },
-    "summarization_wikihow": {
-      "length_penalty": 0.6,
-      "max_length": 256,
-      "max_position_embeddings": 512
-    },
-    "summarization_xsum": {
-      "length_penalty": 0.8,
-      "max_length": 64,
-      "max_position_embeddings": 512
-    }
-  },
   "torch_dtype": "float32",
   "transformers_version": "4.35.2",
   "use_cache": true,

 {
+  "_name_or_path": "google/pegasus-cnn_dailymail",
   "activation_dropout": 0.1,
   "activation_function": "relu",
   "add_bias_logits": false,
   "attention_dropout": 0.1,
   "bos_token_id": 0,
   "classif_dropout": 0.0,
   "d_model": 1024,
   "decoder_attention_heads": 16,
   "decoder_ffn_dim": 4096,
   "encoder_layers": 16,
   "eos_token_id": 1,
   "extra_pos_embeddings": 1,
   "forced_eos_token_id": 1,
   "id2label": {
     "0": "LABEL_0",
     "1": "LABEL_1",
     "LABEL_2": 2
   },
   "length_penalty": 0.8,
+  "max_length": 128,
   "max_position_embeddings": 1024,
+  "min_length": 32,
   "model_type": "pegasus",
   "normalize_before": true,
   "normalize_embedding": false,
   "pad_token_id": 0,
   "scale_embedding": true,
   "static_position_embeddings": true,
   "torch_dtype": "float32",
   "transformers_version": "4.35.2",
   "use_cache": true,

checkpoint-1000/generation_config.json CHANGED Viewed

@@ -4,7 +4,8 @@
   "eos_token_id": 1,
   "forced_eos_token_id": 1,
   "length_penalty": 0.8,
-  "max_length": 256,
   "num_beams": 8,
   "pad_token_id": 0,
   "transformers_version": "4.35.2"

   "eos_token_id": 1,
   "forced_eos_token_id": 1,
   "length_penalty": 0.8,
+  "max_length": 128,
+  "min_length": 32,
   "num_beams": 8,
   "pad_token_id": 0,
   "transformers_version": "4.35.2"

checkpoint-1000/model.safetensors CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:19d44606646abedfeff7cf7c6ee2a06921e1022af697cd6a9e3390391a155ac3
 size 2283652852

 version https://git-lfs.github.com/spec/v1
+oid sha256:fc975443e2eecbee388630f54f420bcec81737434bc9d6d46b0561f601e67572
 size 2283652852

checkpoint-1000/optimizer.pt CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:cc2e5a49d6f16090a3cedb1222c08610acc19fc48c0fa7595ed341cb53b9d87f
 size 4550170737

 version https://git-lfs.github.com/spec/v1
+oid sha256:f126fb78514701ee7ab1e8251fae95a6fb208e28cd53f0ed8d7407bdb2451fc9
 size 4550170737

checkpoint-1000/rng_state.pth CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:24331779cf5a879ef2d67d98058afca3fe15b1b15824c82e8dc22985f06fde1c
 size 14244

 version https://git-lfs.github.com/spec/v1
+oid sha256:39321034682cdf62b8b5f1b226a7deb62ebf0e5e619ffc243c1f9bf89df4089b
 size 14244

checkpoint-1000/trainer_state.json CHANGED Viewed

@@ -10,60 +10,60 @@
   "log_history": [
     {
       "epoch": 1.0,
-      "eval_loss": 1.7080590724945068,
       "eval_rouge-1": {
-        "f": 0.3602114751016854,
-        "p": 0.34496959357843615,
-        "r": 0.4226024767202879
       },
       "eval_rouge-2": {
-        "f": 0.15643009303399508,
-        "p": 0.15275740366063076,
-        "r": 0.19066724740497631
       },
       "eval_rouge-l": {
-        "f": 0.32208903782335413,
-        "p": 0.30833161329969466,
-        "r": 0.37836219852997405
       },
-      "eval_runtime": 2219.4171,
-      "eval_samples_per_second": 0.382,
-      "eval_steps_per_second": 0.048,
       "step": 423
     },
     {
       "epoch": 1.18,
       "learning_rate": 6.1150512214342e-07,
-      "loss": 2.1563,
       "step": 500
     },
     {
       "epoch": 2.0,
-      "eval_loss": 1.6040149927139282,
       "eval_rouge-1": {
-        "f": 0.3612603719250026,
-        "p": 0.34358610140933893,
-        "r": 0.42870738318419865
       },
       "eval_rouge-2": {
-        "f": 0.157552839681331,
-        "p": 0.15225476322754022,
-        "r": 0.19552963369178558
       },
       "eval_rouge-l": {
-        "f": 0.3229502551530144,
-        "p": 0.30691966907819435,
-        "r": 0.3838586158422521
       },
-      "eval_runtime": 2223.1583,
-      "eval_samples_per_second": 0.381,
-      "eval_steps_per_second": 0.048,
       "step": 846
     },
     {
       "epoch": 2.36,
       "learning_rate": 2.1907013396375099e-07,
-      "loss": 1.8944,
       "step": 1000
     }
   ],
@@ -71,7 +71,7 @@
   "max_steps": 1269,
   "num_train_epochs": 3,
   "save_steps": 500,
-  "total_flos": 1.1508669039968256e+16,
   "trial_name": null,
   "trial_params": null
 }

   "log_history": [
     {
       "epoch": 1.0,
+      "eval_loss": 2.069897174835205,
       "eval_rouge-1": {
+        "f": 0.3774477031775325,
+        "p": 0.3995562278312168,
+        "r": 0.3726788874248043
       },
       "eval_rouge-2": {
+        "f": 0.17076277482918784,
+        "p": 0.18355330883104226,
+        "r": 0.16801465382058134
       },
       "eval_rouge-l": {
+        "f": 0.3426353352122299,
+        "p": 0.3630648264869712,
+        "r": 0.33789919649146194
       },
+      "eval_runtime": 951.0251,
+      "eval_samples_per_second": 0.891,
+      "eval_steps_per_second": 0.111,
       "step": 423
     },
     {
       "epoch": 1.18,
       "learning_rate": 6.1150512214342e-07,
+      "loss": 2.7455,
       "step": 500
     },
     {
       "epoch": 2.0,
+      "eval_loss": 1.9773768186569214,
       "eval_rouge-1": {
+        "f": 0.38869549773455014,
+        "p": 0.4028061100223781,
+        "r": 0.39196258894703817
       },
       "eval_rouge-2": {
+        "f": 0.18064299244610635,
+        "p": 0.18994216402253805,
+        "r": 0.18189618469586705
       },
       "eval_rouge-l": {
+        "f": 0.3535216132968263,
+        "p": 0.3670341581740442,
+        "r": 0.3558599440954195
       },
+      "eval_runtime": 966.3939,
+      "eval_samples_per_second": 0.876,
+      "eval_steps_per_second": 0.11,
       "step": 846
     },
     {
       "epoch": 2.36,
       "learning_rate": 2.1907013396375099e-07,
+      "loss": 2.5092,
       "step": 1000
     }
   ],
   "max_steps": 1269,
   "num_train_epochs": 3,
   "save_steps": 500,
+  "total_flos": 1.1514312525152256e+16,
   "trial_name": null,
   "trial_params": null
 }

checkpoint-1000/training_args.bin CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:516e059b7c8633fc443edc3f49f50ac8c78fe9a0ca1e274222be68a0dfd4a5b9
 size 4728

 version https://git-lfs.github.com/spec/v1
+oid sha256:d857e9181d05ce6c81a8a974a12df0eabe666a7930c097557fcfb2f44cbb2e1e
 size 4728

checkpoint-500/config.json CHANGED Viewed

@@ -1,5 +1,5 @@
 {
-  "_name_or_path": "google/pegasus-large",
   "activation_dropout": 0.1,
   "activation_function": "relu",
   "add_bias_logits": false,
@@ -10,7 +10,6 @@
   "attention_dropout": 0.1,
   "bos_token_id": 0,
   "classif_dropout": 0.0,
-  "classifier_dropout": 0.0,
   "d_model": 1024,
   "decoder_attention_heads": 16,
   "decoder_ffn_dim": 4096,
@@ -24,9 +23,7 @@
   "encoder_layers": 16,
   "eos_token_id": 1,
   "extra_pos_embeddings": 1,
-  "force_bos_token_to_be_generated": false,
   "forced_eos_token_id": 1,
-  "gradient_checkpointing": false,
   "id2label": {
     "0": "LABEL_0",
     "1": "LABEL_1",
@@ -40,8 +37,9 @@
     "LABEL_2": 2
   },
   "length_penalty": 0.8,
-  "max_length": 256,
   "max_position_embeddings": 1024,
   "model_type": "pegasus",
   "normalize_before": true,
   "normalize_embedding": false,
@@ -50,73 +48,6 @@
   "pad_token_id": 0,
   "scale_embedding": true,
   "static_position_embeddings": true,
-  "task_specific_params": {
-    "summarization_aeslc": {
-      "length_penalty": 0.6,
-      "max_length": 32,
-      "max_position_embeddings": 512
-    },
-    "summarization_arxiv": {
-      "length_penalty": 0.8,
-      "max_length": 256,
-      "max_position_embeddings": 1024
-    },
-    "summarization_big_patent": {
-      "length_penalty": 0.7,
-      "max_length": 256,
-      "max_position_embeddings": 1024
-    },
-    "summarization_billsum": {
-      "length_penalty": 0.6,
-      "max_length": 256,
-      "max_position_embeddings": 1024
-    },
-    "summarization_cnn_dailymail": {
-      "length_penalty": 0.8,
-      "max_length": 128,
-      "max_position_embeddings": 1024
-    },
-    "summarization_gigaword": {
-      "length_penalty": 0.6,
-      "max_length": 32,
-      "max_position_embeddings": 128
-    },
-    "summarization_large": {
-      "length_penalty": 0.8,
-      "max_length": 256,
-      "max_position_embeddings": 1024
-    },
-    "summarization_multi_news": {
-      "length_penalty": 0.8,
-      "max_length": 256,
-      "max_position_embeddings": 1024
-    },
-    "summarization_newsroom": {
-      "length_penalty": 0.8,
-      "max_length": 128,
-      "max_position_embeddings": 512
-    },
-    "summarization_pubmed": {
-      "length_penalty": 0.8,
-      "max_length": 256,
-      "max_position_embeddings": 1024
-    },
-    "summarization_reddit_tifu": {
-      "length_penalty": 0.6,
-      "max_length": 128,
-      "max_position_embeddings": 512
-    },
-    "summarization_wikihow": {
-      "length_penalty": 0.6,
-      "max_length": 256,
-      "max_position_embeddings": 512
-    },
-    "summarization_xsum": {
-      "length_penalty": 0.8,
-      "max_length": 64,
-      "max_position_embeddings": 512
-    }
-  },
   "torch_dtype": "float32",
   "transformers_version": "4.35.2",
   "use_cache": true,

 {
+  "_name_or_path": "google/pegasus-cnn_dailymail",
   "activation_dropout": 0.1,
   "activation_function": "relu",
   "add_bias_logits": false,
   "attention_dropout": 0.1,
   "bos_token_id": 0,
   "classif_dropout": 0.0,
   "d_model": 1024,
   "decoder_attention_heads": 16,
   "decoder_ffn_dim": 4096,
   "encoder_layers": 16,
   "eos_token_id": 1,
   "extra_pos_embeddings": 1,
   "forced_eos_token_id": 1,
   "id2label": {
     "0": "LABEL_0",
     "1": "LABEL_1",
     "LABEL_2": 2
   },
   "length_penalty": 0.8,
+  "max_length": 128,
   "max_position_embeddings": 1024,
+  "min_length": 32,
   "model_type": "pegasus",
   "normalize_before": true,
   "normalize_embedding": false,
   "pad_token_id": 0,
   "scale_embedding": true,
   "static_position_embeddings": true,
   "torch_dtype": "float32",
   "transformers_version": "4.35.2",
   "use_cache": true,

checkpoint-500/generation_config.json CHANGED Viewed

@@ -4,7 +4,8 @@
   "eos_token_id": 1,
   "forced_eos_token_id": 1,
   "length_penalty": 0.8,
-  "max_length": 256,
   "num_beams": 8,
   "pad_token_id": 0,
   "transformers_version": "4.35.2"

   "eos_token_id": 1,
   "forced_eos_token_id": 1,
   "length_penalty": 0.8,
+  "max_length": 128,
+  "min_length": 32,
   "num_beams": 8,
   "pad_token_id": 0,
   "transformers_version": "4.35.2"

checkpoint-500/model.safetensors CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:82cd843244eafc2ef804646426b877107fc55ef0e31d9e7f6f91e2d6ddd235e1
 size 2283652852

 version https://git-lfs.github.com/spec/v1
+oid sha256:5a55fc9a51b4dafd4e2da8a9991ea57100024c2756c6b76601c0479b20aef950
 size 2283652852

checkpoint-500/optimizer.pt CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:22641152bbf9588c1cdef6a0daf16c4fa0ad61683710bacf7347774f42c547b2
 size 4550170737

 version https://git-lfs.github.com/spec/v1
+oid sha256:c847315530b20a629e7913e40956f0bd8201ba86e1acb6d34649be925603d40c
 size 4550170737

checkpoint-500/rng_state.pth CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:720e736989b19f04509323a646e0f7ac91dd5fe6920d5a542aaa26a2b8f738ac
 size 14244

 version https://git-lfs.github.com/spec/v1
+oid sha256:6e19a3fa6b1bb8fe344226cf1d6a69b5fa2f0142cb55e74a7738945e2b10eafd
 size 14244

checkpoint-500/trainer_state.json CHANGED Viewed

@@ -10,31 +10,31 @@
   "log_history": [
     {
       "epoch": 1.0,
-      "eval_loss": 1.7080590724945068,
       "eval_rouge-1": {
-        "f": 0.3602114751016854,
-        "p": 0.34496959357843615,
-        "r": 0.4226024767202879
       },
       "eval_rouge-2": {
-        "f": 0.15643009303399508,
-        "p": 0.15275740366063076,
-        "r": 0.19066724740497631
       },
       "eval_rouge-l": {
-        "f": 0.32208903782335413,
-        "p": 0.30833161329969466,
-        "r": 0.37836219852997405
       },
-      "eval_runtime": 2219.4171,
-      "eval_samples_per_second": 0.382,
-      "eval_steps_per_second": 0.048,
       "step": 423
     },
     {
       "epoch": 1.18,
       "learning_rate": 6.1150512214342e-07,
-      "loss": 2.1563,
       "step": 500
     }
   ],
@@ -42,7 +42,7 @@
   "max_steps": 1269,
   "num_train_epochs": 3,
   "save_steps": 500,
-  "total_flos": 5752472169873408.0,
   "trial_name": null,
   "trial_params": null
 }

   "log_history": [
     {
       "epoch": 1.0,
+      "eval_loss": 2.069897174835205,
       "eval_rouge-1": {
+        "f": 0.3774477031775325,
+        "p": 0.3995562278312168,
+        "r": 0.3726788874248043
       },
       "eval_rouge-2": {
+        "f": 0.17076277482918784,
+        "p": 0.18355330883104226,
+        "r": 0.16801465382058134
       },
       "eval_rouge-l": {
+        "f": 0.3426353352122299,
+        "p": 0.3630648264869712,
+        "r": 0.33789919649146194
       },
+      "eval_runtime": 951.0251,
+      "eval_samples_per_second": 0.891,
+      "eval_steps_per_second": 0.111,
       "step": 423
     },
     {
       "epoch": 1.18,
       "learning_rate": 6.1150512214342e-07,
+      "loss": 2.7455,
       "step": 500
     }
   ],
   "max_steps": 1269,
   "num_train_epochs": 3,
   "save_steps": 500,
+  "total_flos": 5756986958020608.0,
   "trial_name": null,
   "trial_params": null
 }

checkpoint-500/training_args.bin CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:516e059b7c8633fc443edc3f49f50ac8c78fe9a0ca1e274222be68a0dfd4a5b9
 size 4728

 version https://git-lfs.github.com/spec/v1
+oid sha256:d857e9181d05ce6c81a8a974a12df0eabe666a7930c097557fcfb2f44cbb2e1e
 size 4728

config.json CHANGED Viewed

@@ -1,5 +1,5 @@
 {
-  "_name_or_path": "google/pegasus-large",
   "activation_dropout": 0.1,
   "activation_function": "relu",
   "add_bias_logits": false,
@@ -10,7 +10,6 @@
   "attention_dropout": 0.1,
   "bos_token_id": 0,
   "classif_dropout": 0.0,
-  "classifier_dropout": 0.0,
   "d_model": 1024,
   "decoder_attention_heads": 16,
   "decoder_ffn_dim": 4096,
@@ -24,9 +23,7 @@
   "encoder_layers": 16,
   "eos_token_id": 1,
   "extra_pos_embeddings": 1,
-  "force_bos_token_to_be_generated": false,
   "forced_eos_token_id": 1,
-  "gradient_checkpointing": false,
   "id2label": {
     "0": "LABEL_0",
     "1": "LABEL_1",
@@ -40,8 +37,9 @@
     "LABEL_2": 2
   },
   "length_penalty": 0.8,
-  "max_length": 256,
   "max_position_embeddings": 1024,
   "model_type": "pegasus",
   "normalize_before": true,
   "normalize_embedding": false,
@@ -50,73 +48,6 @@
   "pad_token_id": 0,
   "scale_embedding": true,
   "static_position_embeddings": true,
-  "task_specific_params": {
-    "summarization_aeslc": {
-      "length_penalty": 0.6,
-      "max_length": 32,
-      "max_position_embeddings": 512
-    },
-    "summarization_arxiv": {
-      "length_penalty": 0.8,
-      "max_length": 256,
-      "max_position_embeddings": 1024
-    },
-    "summarization_big_patent": {
-      "length_penalty": 0.7,
-      "max_length": 256,
-      "max_position_embeddings": 1024
-    },
-    "summarization_billsum": {
-      "length_penalty": 0.6,
-      "max_length": 256,
-      "max_position_embeddings": 1024
-    },
-    "summarization_cnn_dailymail": {
-      "length_penalty": 0.8,
-      "max_length": 128,
-      "max_position_embeddings": 1024
-    },
-    "summarization_gigaword": {
-      "length_penalty": 0.6,
-      "max_length": 32,
-      "max_position_embeddings": 128
-    },
-    "summarization_large": {
-      "length_penalty": 0.8,
-      "max_length": 256,
-      "max_position_embeddings": 1024
-    },
-    "summarization_multi_news": {
-      "length_penalty": 0.8,
-      "max_length": 256,
-      "max_position_embeddings": 1024
-    },
-    "summarization_newsroom": {
-      "length_penalty": 0.8,
-      "max_length": 128,
-      "max_position_embeddings": 512
-    },
-    "summarization_pubmed": {
-      "length_penalty": 0.8,
-      "max_length": 256,
-      "max_position_embeddings": 1024
-    },
-    "summarization_reddit_tifu": {
-      "length_penalty": 0.6,
-      "max_length": 128,
-      "max_position_embeddings": 512
-    },
-    "summarization_wikihow": {
-      "length_penalty": 0.6,
-      "max_length": 256,
-      "max_position_embeddings": 512
-    },
-    "summarization_xsum": {
-      "length_penalty": 0.8,
-      "max_length": 64,
-      "max_position_embeddings": 512
-    }
-  },
   "torch_dtype": "float32",
   "transformers_version": "4.35.2",
   "use_cache": true,

 {
+  "_name_or_path": "google/pegasus-cnn_dailymail",
   "activation_dropout": 0.1,
   "activation_function": "relu",
   "add_bias_logits": false,
   "attention_dropout": 0.1,
   "bos_token_id": 0,
   "classif_dropout": 0.0,
   "d_model": 1024,
   "decoder_attention_heads": 16,
   "decoder_ffn_dim": 4096,
   "encoder_layers": 16,
   "eos_token_id": 1,
   "extra_pos_embeddings": 1,
   "forced_eos_token_id": 1,
   "id2label": {
     "0": "LABEL_0",
     "1": "LABEL_1",
     "LABEL_2": 2
   },
   "length_penalty": 0.8,
+  "max_length": 128,
   "max_position_embeddings": 1024,
+  "min_length": 32,
   "model_type": "pegasus",
   "normalize_before": true,
   "normalize_embedding": false,
   "pad_token_id": 0,
   "scale_embedding": true,
   "static_position_embeddings": true,
   "torch_dtype": "float32",
   "transformers_version": "4.35.2",
   "use_cache": true,

generation_config.json CHANGED Viewed

@@ -4,7 +4,8 @@
   "eos_token_id": 1,
   "forced_eos_token_id": 1,
   "length_penalty": 0.8,
-  "max_length": 256,
   "num_beams": 8,
   "pad_token_id": 0,
   "transformers_version": "4.35.2"

   "eos_token_id": 1,
   "forced_eos_token_id": 1,
   "length_penalty": 0.8,
+  "max_length": 128,
+  "min_length": 32,
   "num_beams": 8,
   "pad_token_id": 0,
   "transformers_version": "4.35.2"

model.safetensors CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:595c05347075430c6ca1de7f3cbd98a61c95d9304b9ebe973f17a362d92bca01
 size 2283652852

 version https://git-lfs.github.com/spec/v1
+oid sha256:309daf1e4a09dbad57538ba41a24416a4390f5b8244a0477986e0b2140029b93
 size 2283652852

training_args.bin CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:516e059b7c8633fc443edc3f49f50ac8c78fe9a0ca1e274222be68a0dfd4a5b9
 size 4728

 version https://git-lfs.github.com/spec/v1
+oid sha256:d857e9181d05ce6c81a8a974a12df0eabe666a7930c097557fcfb2f44cbb2e1e
 size 4728