emfomy commited on
Commit
c75a5c5
β€’
1 Parent(s): 1112d4e

Upload model files.

Browse files
README.md ADDED
@@ -0,0 +1,22 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ language:
3
+ - zh
4
+ thumbnail: https://ckip.iis.sinica.edu.tw/files/ckip_logo.png
5
+ tags:
6
+ - pytorch
7
+ - token-classification
8
+ - albert
9
+ - zh
10
+ license: gpl-3.0
11
+ datasets:
12
+ metrics:
13
+ ---
14
+
15
+ # CKIP ALBERT Tiny Chinese β€” Word Segmentation
16
+
17
+ ## Contributers
18
+
19
+ * [Mu Yang](https://muyang.pro) at [CKIP](https://ckip.iis.sinica.edu.tw) (Author & Maintainer)
20
+
21
+ ## Attention
22
+ Please Use `BertTokenizer` instead of `AutoTokenizer`!!!
config.json ADDED
@@ -0,0 +1,37 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "AlbertForTokenClassification"
4
+ ],
5
+ "attention_probs_dropout_prob": 0.0,
6
+ "bos_token_id": 2,
7
+ "classifier_dropout_prob": 0.1,
8
+ "down_scale_factor": 1,
9
+ "embedding_size": 128,
10
+ "eos_token_id": 3,
11
+ "gap_size": 0,
12
+ "hidden_act": "gelu",
13
+ "hidden_dropout_prob": 0.0,
14
+ "hidden_size": 312,
15
+ "id2label": {
16
+ "0": "B",
17
+ "1": "I"
18
+ },
19
+ "initializer_range": 0.02,
20
+ "inner_group_num": 1,
21
+ "intermediate_size": 1248,
22
+ "label2id": {
23
+ "B": 0,
24
+ "I": 1
25
+ },
26
+ "layer_norm_eps": 1e-12,
27
+ "max_position_embeddings": 512,
28
+ "model_type": "albert",
29
+ "net_structure_type": 0,
30
+ "num_attention_heads": 12,
31
+ "num_hidden_groups": 1,
32
+ "num_hidden_layers": 4,
33
+ "num_memory_blocks": 0,
34
+ "pad_token_id": 0,
35
+ "type_vocab_size": 2,
36
+ "vocab_size": 21128
37
+ }
pytorch_model.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2ededb20956722e26434fcda12b31a1da568e1629228be273cffe53f275b9d94
3
+ size 15949017
special_tokens_map.json ADDED
@@ -0,0 +1 @@
 
1
+ {"unk_token": "[UNK]", "sep_token": "[SEP]", "pad_token": "[PAD]", "cls_token": "[CLS]", "mask_token": "[MASK]"}
tokenizer_config.json ADDED
@@ -0,0 +1 @@
 
1
+ {"do_lower_case": false, "do_basic_tokenize": true, "never_split": null, "unk_token": "[UNK]", "sep_token": "[SEP]", "pad_token": "[PAD]", "cls_token": "[CLS]", "mask_token": "[MASK]", "tokenize_chinese_chars": true, "strip_accents": null, "model_max_length": 512, "name_or_path": "bert-base-chinese"}
vocab.txt ADDED
The diff for this file is too large to render. See raw diff