Training in progress epoch 0

Files changed (7) hide show

README.md CHANGED Viewed

@@ -12,10 +12,10 @@ probably proofread and complete it, then remove this comment. -->
 # hm192494/distilgpt2-finetuned-wikitext2
-This model is a fine-tuned version of [distilgpt2](https://huggingface.co/distilgpt2) on an unknown dataset.
 It achieves the following results on the evaluation set:
-- Train Loss: 3.3277
-- Validation Loss: 3.7689
 - Epoch: 0
 ## Model description
@@ -42,7 +42,7 @@ The following hyperparameters were used during training:
 | Train Loss | Validation Loss | Epoch |
 |:----------:|:---------------:|:-----:|
-| 3.3277     | 3.7689          | 0     |
 ### Framework versions

 # hm192494/distilgpt2-finetuned-wikitext2
+This model is a fine-tuned version of [distilroberta-base](https://huggingface.co/distilroberta-base) on an unknown dataset.
 It achieves the following results on the evaluation set:
+- Train Loss: 6.1414
+- Validation Loss: 5.4539
 - Epoch: 0
 ## Model description
 | Train Loss | Validation Loss | Epoch |
 |:----------:|:---------------:|:-----:|
+| 6.1414     | 5.4539          | 0     |
 ### Framework versions

config.json CHANGED Viewed

@@ -1,45 +1,26 @@
 {
-  "_name_or_path": "distilgpt2",
-  "_num_labels": 1,
-  "activation_function": "gelu_new",
   "architectures": [
-    "GPT2LMHeadModel"
   ],
-  "attn_pdrop": 0.1,
-  "bos_token_id": 50256,
-  "embd_pdrop": 0.1,
-  "eos_token_id": 50256,
-  "id2label": {
-    "0": "LABEL_0"
-  },
   "initializer_range": 0.02,
-  "label2id": {
-    "LABEL_0": 0
-  },
-  "layer_norm_epsilon": 1e-05,
-  "model_type": "gpt2",
-  "n_ctx": 1024,
-  "n_embd": 768,
-  "n_head": 12,
-  "n_inner": null,
-  "n_layer": 6,
-  "n_positions": 1024,
-  "reorder_and_upcast_attn": false,
-  "resid_pdrop": 0.1,
-  "scale_attn_by_inverse_layer_idx": false,
-  "scale_attn_weights": true,
-  "summary_activation": null,
-  "summary_first_dropout": 0.1,
-  "summary_proj_to_labels": true,
-  "summary_type": "cls_index",
-  "summary_use_proj": true,
-  "task_specific_params": {
-    "text-generation": {
-      "do_sample": true,
-      "max_length": 50
-    }
-  },
   "transformers_version": "4.19.1",
   "use_cache": true,
-  "vocab_size": 50257
 }

 {
+  "_name_or_path": "distilroberta-base",
   "architectures": [
+    "RobertaForMaskedLM"
   ],
+  "attention_probs_dropout_prob": 0.1,
+  "bos_token_id": 0,
+  "classifier_dropout": null,
+  "eos_token_id": 2,
+  "hidden_act": "gelu",
+  "hidden_dropout_prob": 0.1,
+  "hidden_size": 768,
   "initializer_range": 0.02,
+  "intermediate_size": 3072,
+  "layer_norm_eps": 1e-05,
+  "max_position_embeddings": 514,
+  "model_type": "roberta",
+  "num_attention_heads": 12,
+  "num_hidden_layers": 6,
+  "pad_token_id": 1,
+  "position_embedding_type": "absolute",
   "transformers_version": "4.19.1",
+  "type_vocab_size": 1,
   "use_cache": true,
+  "vocab_size": 50265
 }

special_tokens_map.json CHANGED Viewed

	@@ -1 +1 @@
1	- {"bos_token": "~~<\|endoftext\|>~~", "eos_token": "~~<\|endoftext\|>~~", "unk_token": "~~<\|endoftext\|>~~"}


1	+ {"bos_token": "<s>", "eos_token": "</s>", "unk_token": "<unk>", "sep_token": "</s>", "pad_token": "<pad>", "cls_token": "<s>", "mask_token": {"content": "<mask>", "single_word": false, "lstrip": true, "rstrip": false, "normalized": false}}

tf_model.h5 CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:5eef0d6b2075845ecf9a3ceb1a92cb04ca0b9b66d7f29f27b71cb506f0d95b41
-size 327745496

 version https://git-lfs.github.com/spec/v1
+oid sha256:07443bf6fd58aeef7a2204b225bbb6eeb38a434fef54f83e7881a0c38aa5544f
+size 484838716

tokenizer.json CHANGED Viewed

The diff for this file is too large to render. See raw diff

tokenizer_config.json CHANGED Viewed

	@@ -1 +1 @@
1	- {"~~unk_token~~": "~~<\|endoftext\|>~~", "bos_token": "~~<\|endoftext\|>~~", "eos_token": "~~<\|endoftext\|>~~", "add_prefix_space": false, "model_max_length": ~~1024~~, "special_tokens_map_file": null, "name_or_path": "~~distilgpt2~~", "tokenizer_class": "~~GPT2Tokenizer~~"}


1	+ {"errors": "replace", "bos_token": "<s>", "eos_token": "</s>", "sep_token": "</s>", "cls_token": "<s>", "unk_token": "<unk>", "pad_token": "<pad>", "mask_token": "<mask>", "add_prefix_space": false, "trim_offsets": true, "model_max_length": 512, "special_tokens_map_file": null, "name_or_path": "distilroberta-base", "tokenizer_class": "RobertaTokenizer"}

vocab.json CHANGED Viewed

The diff for this file is too large to render. See raw diff