Add SetFit ABSA model

Browse files

Files changed (14) hide show

.gitattributes +1 -0
1_Pooling/config.json +3 -3
README.md +102 -92
config.json +14 -33
config_sentence_transformers.json +2 -2
config_setfit.json +2 -2
model.safetensors +2 -2
model_head.pkl +2 -2
modules.json +6 -0
sentence_bert_config.json +1 -1
sentencepiece.bpe.model +3 -0
special_tokens_map.json +20 -6
tokenizer.json +0 -0
tokenizer_config.json +18 -20

.gitattributes CHANGED Viewed

@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text

 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text
+tokenizer.json filter=lfs diff=lfs merge=lfs -text

1_Pooling/config.json CHANGED Viewed

@@ -1,7 +1,7 @@
 {
-  "word_embedding_dimension": 768,
-  "pooling_mode_cls_token": false,
-  "pooling_mode_mean_tokens": true,
   "pooling_mode_max_tokens": false,
   "pooling_mode_mean_sqrt_len_tokens": false
 }

 {
+  "word_embedding_dimension": 1024,
+  "pooling_mode_cls_token": true,
+  "pooling_mode_mean_tokens": false,
   "pooling_mode_max_tokens": false,
   "pooling_mode_mean_sqrt_len_tokens": false
 }

README.md CHANGED Viewed

@@ -9,22 +9,22 @@ tags:
 metrics:
 - accuracy
 widget:
-- text: di area tersebut makanan perancis:mungkin agak ramai di akhir pekan, tapi
-    suasana bagus dan ini adalah makanan prancis terbaik yang bisa anda temukan di
-    area tersebut makanan perancis
-- text: para pelayan dan pemilik tidak:para pelayan dan pemilik tidak peduli tentang
-    hal ini dan berjanji untuk memanggil pembasmi tetapi tidak kecewa atau meminta
-    maaf seperti yang saya harapkan.
-- text: suasana ramai tapi suasana:suasana ramai tapi suasana seperti bistro.
-- text: menyukai artisanal! tarif perancis:jika anda menyukai anggur dan keju serta
-    hidangan prancis yang lezat, anda akan menyukai artisanal! tarif perancis
-- text: hebat lagi, harga juga sangat terjangkau:hebat lagi, harga juga sangat terjangkau,
-    dan makanan sangat enak.
 pipeline_tag: text-classification
 inference: false
-base_model: firqaaa/indo-sentence-bert-base
 model-index:
-- name: SetFit Polarity Model with firqaaa/indo-sentence-bert-base
   results:
   - task:
       type: text-classification
@@ -35,13 +35,13 @@ model-index:
       split: test
     metrics:
     - type: accuracy
-      value: 0.6836734693877551
       name: Accuracy
 ---
-# SetFit Polarity Model with firqaaa/indo-sentence-bert-base
-This is a [SetFit](https://github.com/huggingface/setfit) model that can be used for Aspect Based Sentiment Analysis (ABSA). This SetFit model uses [firqaaa/indo-sentence-bert-base](https://huggingface.co/firqaaa/indo-sentence-bert-base) as the Sentence Transformer embedding model. A [LogisticRegression](https://scikit-learn.org/stable/modules/generated/sklearn.linear_model.LogisticRegression.html) instance is used for classification. In particular, this model is in charge of classifying aspect polarities.
 The model has been trained using an efficient few-shot learning technique that involves:
@@ -58,12 +58,12 @@ This model was trained within the context of a larger system for ABSA, which loo
 ### Model Description
 - **Model Type:** SetFit
-- **Sentence Transformer body:** [firqaaa/indo-sentence-bert-base](https://huggingface.co/firqaaa/indo-sentence-bert-base)
 - **Classification head:** a [LogisticRegression](https://scikit-learn.org/stable/modules/generated/sklearn.linear_model.LogisticRegression.html) instance
 - **spaCy Model:** id_core_news_trf
-- **SetFitABSA Aspect Model:** [firqaaa/indo-setfit-absa-sentence-bert-base-p1-restaurants-aspect](https://huggingface.co/firqaaa/indo-setfit-absa-sentence-bert-base-p1-restaurants-aspect)
-- **SetFitABSA Polarity Model:** [firqaaa/indo-setfit-absa-sentence-bert-base-p1-restaurants-polarity](https://huggingface.co/firqaaa/indo-setfit-absa-sentence-bert-base-p1-restaurants-polarity)
-- **Maximum Sequence Length:** 512 tokens
 - **Number of Classes:** 4 classes
 <!-- - **Training Dataset:** [Unknown](https://huggingface.co/datasets/unknown) -->
 <!-- - **Language:** Unknown -->
@@ -88,7 +88,7 @@ This model was trained within the context of a larger system for ABSA, which loo
 ### Metrics
 | Label   | Accuracy |
 |:--------|:---------|
-| **all** | 0.6837   |
 ## Uses
@@ -107,8 +107,8 @@ from setfit import AbsaModel
 # Download from the 🤗 Hub
 model = AbsaModel.from_pretrained(
-    "firqaaa/indo-setfit-absa-sentence-bert-base-p1-restaurants-aspect",
-    "firqaaa/indo-setfit-absa-sentence-bert-base-p1-restaurants-polarity",
 )
 # Run inference
 preds = model("The food was great, but the venue is just way too busy.")
@@ -143,17 +143,17 @@ preds = model("The food was great, but the venue is just way too busy.")
 ### Training Set Metrics
 | Training set | Min | Median  | Max |
 |:-------------|:----|:--------|:----|
-| Word count   | 3   | 20.9935 | 62  |
 | Label   | Training Sample Count |
 |:--------|:----------------------|
-| konflik | 21                    |
-| negatif | 243                   |
-| netral  | 186                   |
-| positif | 626                   |
 ### Training Hyperparameters
-- batch_size: (32, 32)
 - num_epochs: (1, 1)
 - max_steps: -1
 - sampling_strategy: oversampling
@@ -170,69 +170,79 @@ preds = model("The food was great, but the venue is just way too busy.")
 - load_best_model_at_end: True
 ### Training Results
-| Epoch      | Step    | Training Loss | Validation Loss |
-|:----------:|:-------:|:-------------:|:---------------:|
-| 0.0000     | 1       | 0.2996        | -               |
-| 0.0024     | 50      | 0.2488        | -               |
-| 0.0048     | 100     | 0.2636        | -               |
-| 0.0071     | 150     | 0.2544        | -               |
-| 0.0095     | 200     | 0.2036        | -               |
-| 0.0119     | 250     | 0.2002        | -               |
-| 0.0143     | 300     | 0.1723        | -               |
-| 0.0167     | 350     | 0.2112        | -               |
-| 0.0191     | 400     | 0.1655        | -               |
-| 0.0214     | 450     | 0.1559        | -               |
-| **0.0238** | **500** | **0.0602**    | **0.2033**      |
-| 0.0262     | 550     | 0.1047        | -               |
-| 0.0286     | 600     | 0.1228        | -               |
-| 0.0310     | 650     | 0.1152        | -               |
-| 0.0333     | 700     | 0.0444        | -               |
-| 0.0357     | 750     | 0.0479        | -               |
-| 0.0381     | 800     | 0.065         | -               |
-| 0.0405     | 850     | 0.0417        | -               |
-| 0.0429     | 900     | 0.0647        | -               |
-| 0.0452     | 950     | 0.0517        | -               |
-| 0.0476     | 1000    | 0.0433        | 0.2399          |
-| 0.0500     | 1050    | 0.0044        | -               |
-| 0.0524     | 1100    | 0.0241        | -               |
-| 0.0548     | 1150    | 0.0204        | -               |
-| 0.0572     | 1200    | 0.0532        | -               |
-| 0.0595     | 1250    | 0.0116        | -               |
-| 0.0619     | 1300    | 0.0288        | -               |
-| 0.0643     | 1350    | 0.0125        | -               |
-| 0.0667     | 1400    | 0.0357        | -               |
-| 0.0691     | 1450    | 0.0028        | -               |
-| 0.0714     | 1500    | 0.027         | 0.2564          |
-| 0.0738     | 1550    | 0.0032        | -               |
-| 0.0762     | 1600    | 0.0048        | -               |
-| 0.0786     | 1650    | 0.0003        | -               |
-| 0.0810     | 1700    | 0.0008        | -               |
-| 0.0834     | 1750    | 0.0008        | -               |
-| 0.0857     | 1800    | 0.0023        | -               |
-| 0.0881     | 1850    | 0.0003        | -               |
-| 0.0905     | 1900    | 0.0004        | -               |
-| 0.0929     | 1950    | 0.0003        | -               |
-| 0.0953     | 2000    | 0.0054        | 0.2812          |
-| 0.0976     | 2050    | 0.0005        | -               |
-| 0.1000     | 2100    | 0.0006        | -               |
-| 0.1024     | 2150    | 0.0004        | -               |
-| 0.1048     | 2200    | 0.0019        | -               |
-| 0.1072     | 2250    | 0.0007        | -               |
-| 0.1095     | 2300    | 0.0004        | -               |
-| 0.1119     | 2350    | 0.0001        | -               |
-| 0.1143     | 2400    | 0.0004        | -               |
-| 0.1167     | 2450    | 0.0069        | -               |
-| 0.1191     | 2500    | 0.0001        | 0.2845          |
-| 0.1215     | 2550    | 0.0           | -               |
-| 0.1238     | 2600    | 0.0002        | -               |
-| 0.1262     | 2650    | 0.0001        | -               |
-| 0.1286     | 2700    | 0.0109        | -               |
-| 0.1310     | 2750    | 0.0037        | -               |
-| 0.1334     | 2800    | 0.0001        | -               |
-| 0.1357     | 2850    | 0.0001        | -               |
-| 0.1381     | 2900    | 0.0001        | -               |
-| 0.1405     | 2950    | 0.0001        | -               |
-| 0.1429     | 3000    | 0.0001        | 0.2839          |
 * The bold row denotes the saved checkpoint.
 ### Framework Versions

 metrics:
 - accuracy
 widget:
+- text: gulungan biasa menjadi gulungan luar dalam,:dibutuhkan biaya tambahan $2 untuk
+    mengubah gulungan biasa menjadi gulungan luar dalam, tetapi gulungan tersebut
+    berukuran lebih dari tiga kali lipat, dan itu bukan ha dari nasi.
+- text: -a-bagel (baik di:ess-a-bagel (baik di sty-town atau midtown) sejauh ini merupakan
+    bagel terbaik di ny.
+- text: mahal wadah ini pengelola:ketika kami sedang duduk makan makanan di bawah
+    standar, manajer mulai mencaci-maki beberapa karyawan karena meletakkan wadah
+    bumbu yang salah dan menjelaskan kepada mereka betapa mahal wadah ini pengelola
+- text: staf sangat akomodatif.:staf sangat akomodatif.
+- text: layanan luar biasa melayani:makanan india yang enak dan layanan luar biasa
+    melayani
 pipeline_tag: text-classification
 inference: false
+base_model: BAAI/bge-m3
 model-index:
+- name: SetFit Polarity Model with BAAI/bge-m3
   results:
   - task:
       type: text-classification
       split: test
     metrics:
     - type: accuracy
+      value: 0.7898320472083522
       name: Accuracy
 ---
+# SetFit Polarity Model with BAAI/bge-m3
+This is a [SetFit](https://github.com/huggingface/setfit) model that can be used for Aspect Based Sentiment Analysis (ABSA). This SetFit model uses [BAAI/bge-m3](https://huggingface.co/BAAI/bge-m3) as the Sentence Transformer embedding model. A [LogisticRegression](https://scikit-learn.org/stable/modules/generated/sklearn.linear_model.LogisticRegression.html) instance is used for classification. In particular, this model is in charge of classifying aspect polarities.
 The model has been trained using an efficient few-shot learning technique that involves:
 ### Model Description
 - **Model Type:** SetFit
+- **Sentence Transformer body:** [BAAI/bge-m3](https://huggingface.co/BAAI/bge-m3)
 - **Classification head:** a [LogisticRegression](https://scikit-learn.org/stable/modules/generated/sklearn.linear_model.LogisticRegression.html) instance
 - **spaCy Model:** id_core_news_trf
+- **SetFitABSA Aspect Model:** [firqaaa/indo-setfit-absa-bert-base-restaurants-aspect](https://huggingface.co/firqaaa/indo-setfit-absa-bert-base-restaurants-aspect)
+- **SetFitABSA Polarity Model:** [firqaaa/indo-setfit-absa-bert-base-restaurants-polarity](https://huggingface.co/firqaaa/indo-setfit-absa-bert-base-restaurants-polarity)
+- **Maximum Sequence Length:** 8192 tokens
 - **Number of Classes:** 4 classes
 <!-- - **Training Dataset:** [Unknown](https://huggingface.co/datasets/unknown) -->
 <!-- - **Language:** Unknown -->
 ### Metrics
 | Label   | Accuracy |
 |:--------|:---------|
+| **all** | 0.7898   |
 ## Uses
 # Download from the 🤗 Hub
 model = AbsaModel.from_pretrained(
+    "firqaaa/indo-setfit-absa-bert-base-restaurants-aspect",
+    "firqaaa/indo-setfit-absa-bert-base-restaurants-polarity",
 )
 # Run inference
 preds = model("The food was great, but the venue is just way too busy.")
 ### Training Set Metrics
 | Training set | Min | Median  | Max |
 |:-------------|:----|:--------|:----|
+| Word count   | 3   | 20.6594 | 62  |
 | Label   | Training Sample Count |
 |:--------|:----------------------|
+| konflik | 34                    |
+| negatif | 323                   |
+| netral  | 258                   |
+| positif | 853                   |
 ### Training Hyperparameters
+- batch_size: (16, 16)
 - num_epochs: (1, 1)
 - max_steps: -1
 - sampling_strategy: oversampling
 - load_best_model_at_end: True
 ### Training Results
+| Epoch      | Step     | Training Loss | Validation Loss |
+|:----------:|:--------:|:-------------:|:---------------:|
+| 0.0000     | 1        | 0.2345        | -               |
+| 0.0006     | 50       | 0.2337        | -               |
+| 0.0013     | 100      | 0.267         | -               |
+| 0.0019     | 150      | 0.2335        | -               |
+| 0.0025     | 200      | 0.2368        | -               |
+| 0.0032     | 250      | 0.2199        | -               |
+| 0.0038     | 300      | 0.2325        | -               |
+| 0.0045     | 350      | 0.2071        | -               |
+| 0.0051     | 400      | 0.2229        | -               |
+| 0.0057     | 450      | 0.1153        | -               |
+| 0.0064     | 500      | 0.1771        | 0.1846          |
+| 0.0070     | 550      | 0.1612        | -               |
+| 0.0076     | 600      | 0.1487        | -               |
+| 0.0083     | 650      | 0.147         | -               |
+| 0.0089     | 700      | 0.1982        | -               |
+| 0.0096     | 750      | 0.1579        | -               |
+| 0.0102     | 800      | 0.1148        | -               |
+| 0.0108     | 850      | 0.1008        | -               |
+| 0.0115     | 900      | 0.2035        | -               |
+| 0.0121     | 950      | 0.1348        | -               |
+| **0.0127** | **1000** | **0.0974**    | **0.182**       |
+| 0.0134     | 1050     | 0.121         | -               |
+| 0.0140     | 1100     | 0.1949        | -               |
+| 0.0147     | 1150     | 0.2424        | -               |
+| 0.0153     | 1200     | 0.0601        | -               |
+| 0.0159     | 1250     | 0.0968        | -               |
+| 0.0166     | 1300     | 0.0137        | -               |
+| 0.0172     | 1350     | 0.034         | -               |
+| 0.0178     | 1400     | 0.1217        | -               |
+| 0.0185     | 1450     | 0.0454        | -               |
+| 0.0191     | 1500     | 0.0397        | 0.2216          |
+| 0.0198     | 1550     | 0.0226        | -               |
+| 0.0204     | 1600     | 0.0939        | -               |
+| 0.0210     | 1650     | 0.0537        | -               |
+| 0.0217     | 1700     | 0.0566        | -               |
+| 0.0223     | 1750     | 0.162         | -               |
+| 0.0229     | 1800     | 0.0347        | -               |
+| 0.0236     | 1850     | 0.103         | -               |
+| 0.0242     | 1900     | 0.0615        | -               |
+| 0.0249     | 1950     | 0.0589        | -               |
+| 0.0255     | 2000     | 0.1668        | 0.2132          |
+| 0.0261     | 2050     | 0.1809        | -               |
+| 0.0268     | 2100     | 0.0579        | -               |
+| 0.0274     | 2150     | 0.088         | -               |
+| 0.0280     | 2200     | 0.1047        | -               |
+| 0.0287     | 2250     | 0.1255        | -               |
+| 0.0293     | 2300     | 0.0312        | -               |
+| 0.0300     | 2350     | 0.0097        | -               |
+| 0.0306     | 2400     | 0.0973        | -               |
+| 0.0312     | 2450     | 0.0066        | -               |
+| 0.0319     | 2500     | 0.0589        | 0.2591          |
+| 0.0325     | 2550     | 0.0529        | -               |
+| 0.0331     | 2600     | 0.0169        | -               |
+| 0.0338     | 2650     | 0.0455        | -               |
+| 0.0344     | 2700     | 0.0609        | -               |
+| 0.0350     | 2750     | 0.1151        | -               |
+| 0.0357     | 2800     | 0.0031        | -               |
+| 0.0363     | 2850     | 0.0546        | -               |
+| 0.0370     | 2900     | 0.0051        | -               |
+| 0.0376     | 2950     | 0.0679        | -               |
+| 0.0382     | 3000     | 0.0046        | 0.2646          |
+| 0.0389     | 3050     | 0.011         | -               |
+| 0.0395     | 3100     | 0.0701        | -               |
+| 0.0401     | 3150     | 0.0011        | -               |
+| 0.0408     | 3200     | 0.011         | -               |
+| 0.0414     | 3250     | 0.0026        | -               |
+| 0.0421     | 3300     | 0.0027        | -               |
+| 0.0427     | 3350     | 0.0012        | -               |
+| 0.0433     | 3400     | 0.0454        | -               |
+| 0.0440     | 3450     | 0.0011        | -               |
+| 0.0446     | 3500     | 0.0012        | 0.2602          |
 * The bold row denotes the saved checkpoint.
 ### Framework Versions

config.json CHANGED Viewed

@@ -1,47 +1,28 @@
 {
-  "_name_or_path": "models/step_500/",
-  "_num_labels": 5,
   "architectures": [
-    "BertModel"
   ],
   "attention_probs_dropout_prob": 0.1,
   "classifier_dropout": null,
-  "directionality": "bidi",
   "hidden_act": "gelu",
   "hidden_dropout_prob": 0.1,
-  "hidden_size": 768,
-  "id2label": {
-    "0": "LABEL_0",
-    "1": "LABEL_1",
-    "2": "LABEL_2",
-    "3": "LABEL_3",
-    "4": "LABEL_4"
-  },
   "initializer_range": 0.02,
-  "intermediate_size": 3072,
-  "label2id": {
-    "LABEL_0": 0,
-    "LABEL_1": 1,
-    "LABEL_2": 2,
-    "LABEL_3": 3,
-    "LABEL_4": 4
-  },
-  "layer_norm_eps": 1e-12,
-  "max_position_embeddings": 512,
-  "model_type": "bert",
-  "num_attention_heads": 12,
-  "num_hidden_layers": 12,
   "output_past": true,
-  "pad_token_id": 0,
-  "pooler_fc_size": 768,
-  "pooler_num_attention_heads": 12,
-  "pooler_num_fc_layers": 3,
-  "pooler_size_per_head": 128,
-  "pooler_type": "first_token_transform",
   "position_embedding_type": "absolute",
   "torch_dtype": "float32",
   "transformers_version": "4.36.2",
-  "type_vocab_size": 2,
   "use_cache": true,
-  "vocab_size": 50000
 }

 {
+  "_name_or_path": "models/step_1000/",
   "architectures": [
+    "XLMRobertaModel"
   ],
   "attention_probs_dropout_prob": 0.1,
+  "bos_token_id": 0,
   "classifier_dropout": null,
+  "eos_token_id": 2,
   "hidden_act": "gelu",
   "hidden_dropout_prob": 0.1,
+  "hidden_size": 1024,
   "initializer_range": 0.02,
+  "intermediate_size": 4096,
+  "layer_norm_eps": 1e-05,
+  "max_position_embeddings": 8194,
+  "model_type": "xlm-roberta",
+  "num_attention_heads": 16,
+  "num_hidden_layers": 24,
   "output_past": true,
+  "pad_token_id": 1,
   "position_embedding_type": "absolute",
   "torch_dtype": "float32",
   "transformers_version": "4.36.2",
+  "type_vocab_size": 1,
   "use_cache": true,
+  "vocab_size": 250002
 }

config_sentence_transformers.json CHANGED Viewed

@@ -1,7 +1,7 @@
 {
   "__version__": {
     "sentence_transformers": "2.2.2",
-    "transformers": "4.20.1",
-    "pytorch": "1.11.0"
   }
 }

 {
   "__version__": {
     "sentence_transformers": "2.2.2",
+    "transformers": "4.33.0",
+    "pytorch": "2.1.2+cu121"
   }
 }

config_setfit.json CHANGED Viewed

@@ -1,11 +1,11 @@
 {
   "labels": [
     "konflik",
     "negatif",
     "netral",
     "positif"
   ],
-  "spacy_model": "id_core_news_trf",
-  "span_context": 3,
   "normalize_embeddings": false
 }

 {
+  "span_context": 3,
+  "spacy_model": "id_core_news_trf",
   "labels": [
     "konflik",
     "negatif",
     "netral",
     "positif"
   ],
   "normalize_embeddings": false
 }

model.safetensors CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:9e75903b9d43b34a9602d4470583b192ab954eaad3938be8194373d688bd884c
-size 497787752

 version https://git-lfs.github.com/spec/v1
+oid sha256:7f63dee23a0bf95fda64e8d449f3b986d248f47902b6bd425fe8c3c8a990cf1a
+size 2271064456

model_head.pkl CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:59f44ec07fb9b9dace9551bb690bc911ad26d46608df4ea725d60dc16975f68e
-size 25543

 version https://git-lfs.github.com/spec/v1
+oid sha256:231e6b182373e783eee98fca20da2f2e903c2cd68be2b105c4b08682bb9f33e0
+size 33735

modules.json CHANGED Viewed

@@ -10,5 +10,11 @@
     "name": "1",
     "path": "1_Pooling",
     "type": "sentence_transformers.models.Pooling"
   }
 ]

     "name": "1",
     "path": "1_Pooling",
     "type": "sentence_transformers.models.Pooling"
+  },
+  {
+    "idx": 2,
+    "name": "2",
+    "path": "2_Normalize",
+    "type": "sentence_transformers.models.Normalize"
   }
 ]

sentence_bert_config.json CHANGED Viewed

@@ -1,4 +1,4 @@
 {
-  "max_seq_length": 512,
   "do_lower_case": false
 }

 {
+  "max_seq_length": 8192,
   "do_lower_case": false
 }

sentencepiece.bpe.model ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:cfc8146abe2a0488e9e2a0c56de7952f7c11ab059eca145a0a727afce0db2865
+size 5069051

special_tokens_map.json CHANGED Viewed

@@ -1,34 +1,48 @@
 {
   "cls_token": {
-    "content": "[CLS]",
     "lstrip": false,
     "normalized": false,
     "rstrip": false,
     "single_word": false
   },
-  "mask_token": {
-    "content": "[MASK]",
     "lstrip": false,
     "normalized": false,
     "rstrip": false,
     "single_word": false
   },
   "pad_token": {
-    "content": "[PAD]",
     "lstrip": false,
     "normalized": false,
     "rstrip": false,
     "single_word": false
   },
   "sep_token": {
-    "content": "[SEP]",
     "lstrip": false,
     "normalized": false,
     "rstrip": false,
     "single_word": false
   },
   "unk_token": {
-    "content": "[UNK]",
     "lstrip": false,
     "normalized": false,
     "rstrip": false,

 {
+  "bos_token": {
+    "content": "<s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
   "cls_token": {
+    "content": "<s>",
     "lstrip": false,
     "normalized": false,
     "rstrip": false,
     "single_word": false
   },
+  "eos_token": {
+    "content": "</s>",
     "lstrip": false,
     "normalized": false,
     "rstrip": false,
     "single_word": false
   },
+  "mask_token": {
+    "content": "<mask>",
+    "lstrip": true,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
   "pad_token": {
+    "content": "<pad>",
     "lstrip": false,
     "normalized": false,
     "rstrip": false,
     "single_word": false
   },
   "sep_token": {
+    "content": "</s>",
     "lstrip": false,
     "normalized": false,
     "rstrip": false,
     "single_word": false
   },
   "unk_token": {
+    "content": "<unk>",
     "lstrip": false,
     "normalized": false,
     "rstrip": false,

tokenizer.json CHANGED Viewed

The diff for this file is too large to render. See raw diff

tokenizer_config.json CHANGED Viewed

@@ -1,7 +1,7 @@
 {
   "added_tokens_decoder": {
     "0": {
-      "content": "[PAD]",
       "lstrip": false,
       "normalized": false,
       "rstrip": false,
@@ -9,7 +9,7 @@
       "special": true
     },
     "1": {
-      "content": "[UNK]",
       "lstrip": false,
       "normalized": false,
       "rstrip": false,
@@ -17,7 +17,7 @@
       "special": true
     },
     "2": {
-      "content": "[CLS]",
       "lstrip": false,
       "normalized": false,
       "rstrip": false,
@@ -25,40 +25,38 @@
       "special": true
     },
     "3": {
-      "content": "[SEP]",
       "lstrip": false,
       "normalized": false,
       "rstrip": false,
       "single_word": false,
       "special": true
     },
-    "4": {
-      "content": "[MASK]",
-      "lstrip": false,
       "normalized": false,
       "rstrip": false,
       "single_word": false,
       "special": true
     }
   },
   "clean_up_tokenization_spaces": true,
-  "cls_token": "[CLS]",
-  "do_basic_tokenize": true,
-  "do_lower_case": true,
-  "mask_token": "[MASK]",
-  "max_length": 512,
-  "model_max_length": 1000000000000000019884624838656,
-  "never_split": null,
   "pad_to_multiple_of": null,
-  "pad_token": "[PAD]",
   "pad_token_type_id": 0,
   "padding_side": "right",
-  "sep_token": "[SEP]",
   "stride": 0,
-  "strip_accents": null,
-  "tokenize_chinese_chars": true,
-  "tokenizer_class": "BertTokenizer",
   "truncation_side": "right",
   "truncation_strategy": "longest_first",
-  "unk_token": "[UNK]"
 }

 {
   "added_tokens_decoder": {
     "0": {
+      "content": "<s>",
       "lstrip": false,
       "normalized": false,
       "rstrip": false,
       "special": true
     },
     "1": {
+      "content": "<pad>",
       "lstrip": false,
       "normalized": false,
       "rstrip": false,
       "special": true
     },
     "2": {
+      "content": "</s>",
       "lstrip": false,
       "normalized": false,
       "rstrip": false,
       "special": true
     },
     "3": {
+      "content": "<unk>",
       "lstrip": false,
       "normalized": false,
       "rstrip": false,
       "single_word": false,
       "special": true
     },
+    "250001": {
+      "content": "<mask>",
+      "lstrip": true,
       "normalized": false,
       "rstrip": false,
       "single_word": false,
       "special": true
     }
   },
+  "bos_token": "<s>",
   "clean_up_tokenization_spaces": true,
+  "cls_token": "<s>",
+  "eos_token": "</s>",
+  "mask_token": "<mask>",
+  "max_length": 8192,
+  "model_max_length": 8192,
   "pad_to_multiple_of": null,
+  "pad_token": "<pad>",
   "pad_token_type_id": 0,
   "padding_side": "right",
+  "sep_token": "</s>",
+  "sp_model_kwargs": {},
   "stride": 0,
+  "tokenizer_class": "XLMRobertaTokenizer",
   "truncation_side": "right",
   "truncation_strategy": "longest_first",
+  "unk_token": "<unk>"
 }