added

Browse files

Files changed (9) hide show

README.md +61 -0
added_tokens.json +1 -0
config.json +30 -0
eval.csv +14 -0
pytorch_model.bin +3 -0
special_tokens_map.json +1 -0
spiece.model +3 -0
tokenizer_config.json +1 -0
training_args.bin +3 -0

README.md ADDED Viewed

	@@ -0,0 +1,61 @@

+### Model
+**[`albert-xlarge-v2`](https://huggingface.co/albert-xlarge-v2)** fine-tuned on **[`SQuAD V2`](https://rajpurkar.github.io/SQuAD-explorer/)** using **[`run_squad.py`](https://github.com/huggingface/transformers/blob/master/examples/question-answering/run_squad.py)**
+### Training Parameters
+Trained on 4 NVIDIA GeForce RTX 2080 Ti 11Gb
+```bash
+BASE_MODEL=albert-xlarge-v2
+python run_squad.py \
+  --version_2_with_negative \
+  --model_type albert \
+  --model_name_or_path $BASE_MODEL \
+  --output_dir $OUTPUT_MODEL \
+  --do_eval \
+  --do_lower_case \
+  --train_file $SQUAD_DIR/train-v2.0.json \
+  --predict_file $SQUAD_DIR/dev-v2.0.json \
+  --per_gpu_train_batch_size 3 \
+  --per_gpu_eval_batch_size 64 \
+  --learning_rate 3e-5 \
+  --num_train_epochs 3.0 \
+  --max_seq_length 384 \
+  --doc_stride 128 \
+  --save_steps 2000 \
+  --threads 24 \
+  --warmup_steps 814 \
+  --gradient_accumulation_steps 4 \
+  --fp16 \
+  --do_train
+```
+### Evaluation
+Evaluation on the dev set. I did not sweep for best threshold.
+|                   | val               |
+|-------------------|-------------------|
+| exact             | 84.41842836688285 |
+| f1                | 87.4628460501696  |
+| total             | 11873.0           |
+| HasAns_exact      | 80.68488529014844 |
+| HasAns_f1         | 86.78245127423482 |
+| HasAns_total      | 5928.0            |
+| NoAns_exact       | 88.1412952060555  |
+| NoAns_f1          | 88.1412952060555  |
+| NoAns_total       | 5945.0            |
+| best_exact        | 84.41842836688285 |
+| best_exact_thresh | 0.0               |
+| best_f1           | 87.46284605016956 |
+| best_f1_thresh    | 0.0               |
+### Usage
+See [huggingface documentation](https://huggingface.co/transformers/model_doc/albert.html#albertforquestionanswering). Training on `SQuAD V2` allows the model to score if a paragraph contains an answer:
+```python
+start_scores, end_scores = model(input_ids)
+span_scores = start_scores.softmax(dim=1).log()[:,:,None] + end_scores.softmax(dim=1).log()[:,None,:]
+ignore_score = span_scores[:,0,0] #no answer scores
+```

added_tokens.json ADDED Viewed

	@@ -0,0 +1 @@


1	+ {}

config.json ADDED Viewed

	@@ -0,0 +1,30 @@

+{
+  "architectures": [
+    "AlbertForQuestionAnswering"
+  ],
+  "attention_probs_dropout_prob": 0.1,
+  "bos_token_id": 2,
+  "classifier_dropout_prob": 0.1,
+  "down_scale_factor": 1,
+  "embedding_size": 128,
+  "eos_token_id": 3,
+  "gap_size": 0,
+  "hidden_act": "gelu",
+  "hidden_dropout_prob": 0.1,
+  "hidden_size": 2048,
+  "initializer_range": 0.02,
+  "inner_group_num": 1,
+  "intermediate_size": 8192,
+  "layer_norm_eps": 1e-12,
+  "max_position_embeddings": 512,
+  "model_type": "albert",
+  "net_structure_type": 0,
+  "num_attention_heads": 16,
+  "num_hidden_groups": 1,
+  "num_hidden_layers": 24,
+  "num_memory_blocks": 0,
+  "output_past": true,
+  "pad_token_id": 0,
+  "type_vocab_size": 2,
+  "vocab_size": 30000
+}

eval.csv ADDED Viewed

	@@ -0,0 +1,14 @@

+,val
+exact,84.41842836688285
+f1,87.4628460501696
+total,11873.0
+HasAns_exact,80.68488529014844
+HasAns_f1,86.78245127423482
+HasAns_total,5928.0
+NoAns_exact,88.1412952060555
+NoAns_f1,88.1412952060555
+NoAns_total,5945.0
+best_exact,84.41842836688285
+best_exact_thresh,0.0
+best_f1,87.46284605016956
+best_f1_thresh,0.0

pytorch_model.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:3b759731572f038d3f3d9cb1ef02fac448233dd3d3e4c1b9bfc59e49e87864e5
+size 234922444

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1 @@


1	+ {"bos_token": "[CLS]", "eos_token": "[SEP]", "unk_token": "<unk>", "sep_token": "[SEP]", "pad_token": "<pad>", "cls_token": "[CLS]", "mask_token": "[MASK]"}

spiece.model ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:fefb02b667a6c5c2fe27602d28e5fb3428f66ab89c7d6f388e7c8d44a02d0336
+size 760289

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1 @@


1	+ {"do_lower_case": true, "max_len": 512, "init_inputs": []}

training_args.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:c32566fecd5fffd748f6ab2d71404c2a42d714391eaca4453d3e16c2da226284
+size 1418