ArthurFischel commited on
Commit
0456fe1
1 Parent(s): 4b2bbe8

Upload 10 files

Browse files
.gitattributes CHANGED
@@ -25,7 +25,6 @@
25
  *.safetensors filter=lfs diff=lfs merge=lfs -text
26
  saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
  *.tar.* filter=lfs diff=lfs merge=lfs -text
28
- *.tar filter=lfs diff=lfs merge=lfs -text
29
  *.tflite filter=lfs diff=lfs merge=lfs -text
30
  *.tgz filter=lfs diff=lfs merge=lfs -text
31
  *.wasm filter=lfs diff=lfs merge=lfs -text
 
25
  *.safetensors filter=lfs diff=lfs merge=lfs -text
26
  saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
  *.tar.* filter=lfs diff=lfs merge=lfs -text
 
28
  *.tflite filter=lfs diff=lfs merge=lfs -text
29
  *.tgz filter=lfs diff=lfs merge=lfs -text
30
  *.wasm filter=lfs diff=lfs merge=lfs -text
added_tokens.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ {
2
+ "<image>": 32001
3
+ }
config.json ADDED
@@ -0,0 +1,56 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "additional_vocab_size": 2,
3
+ "alpha_initializer": "ones",
4
+ "alpha_type": "vector",
5
+ "alphas_initializer_range": 0.0,
6
+ "architectures": [
7
+ "IdeficsForVisionText2Text"
8
+ ],
9
+ "bos_token_id": 1,
10
+ "cross_layer_activation_function": "swiglu",
11
+ "cross_layer_interval": 1,
12
+ "dropout": 0.0,
13
+ "eos_token_id": 2,
14
+ "ffn_dim": 64,
15
+ "freeze_lm_head": false,
16
+ "freeze_text_layers": false,
17
+ "freeze_text_module_exceptions": [],
18
+ "freeze_vision_layers": false,
19
+ "freeze_vision_module_exceptions": [],
20
+ "hidden_act": "silu",
21
+ "hidden_size": 16,
22
+ "initializer_range": 0.02,
23
+ "intermediate_size": 11008,
24
+ "max_new_tokens": 128,
25
+ "max_position_embeddings": 128,
26
+ "model_type": "idefics",
27
+ "num_attention_heads": 4,
28
+ "num_hidden_layers": 2,
29
+ "pad_token_id": 0,
30
+ "qk_layer_norms": false,
31
+ "rms_norm_eps": 1e-06,
32
+ "tie_word_embeddings": false,
33
+ "torch_dtype": "float16",
34
+ "transformers_version": "4.27.0.dev0",
35
+ "use_cache": true,
36
+ "use_resampler": true,
37
+ "vocab_size": 32000,
38
+ "word_embed_proj_dim": 16,
39
+ "vision_config": {
40
+ "hidden_act": "gelu",
41
+ "embed_dim": 32,
42
+ "image_size": 30,
43
+ "intermediate_size": 37,
44
+ "patch_size": 2,
45
+ "num_attention_heads": 4,
46
+ "num_hidden_layers": 5,
47
+ "vision_model_name": "hf-internal-testing/tiny-random-clip"
48
+ },
49
+ "perceiver_config": {
50
+ "qk_layer_norms_perceiver": false,
51
+ "resampler_depth": 2,
52
+ "resampler_head_dim": 8,
53
+ "resampler_n_heads": 2,
54
+ "resampler_n_latents": 16
55
+ }
56
+ }
generation_config.json ADDED
@@ -0,0 +1,8 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_from_model_config": true,
3
+ "bos_token_id": 1,
4
+ "eos_token_id": 2,
5
+ "max_new_tokens": 100,
6
+ "pad_token_id": 0,
7
+ "transformers_version": "4.27.0.dev0"
8
+ }
preprocessor_config.json ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "image_num_channels": 3,
3
+ "image_mean": [
4
+ 0.48145466,
5
+ 0.4578275,
6
+ 0.40821073
7
+ ],
8
+ "image_processor_type": "IdeficsImageProcessor",
9
+ "image_size": 30,
10
+ "image_std": [
11
+ 0.26862954,
12
+ 0.26130258,
13
+ 0.27577711
14
+ ],
15
+ "processor_class": "IdeficsProcessor"
16
+ }
pytorch_model.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b6f5de05e77b318f3a80f79e1a8d6b1eda26841ecfd9c0afc4c38a2d14b4e6a5
3
+ size 6467421
special_tokens_map.json ADDED
@@ -0,0 +1,34 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "additional_special_tokens": [
3
+ "<image>",
4
+ "<fake_token_around_image>"
5
+ ],
6
+ "bos_token": {
7
+ "content": "<s>",
8
+ "lstrip": false,
9
+ "normalized": false,
10
+ "rstrip": false,
11
+ "single_word": false
12
+ },
13
+ "eos_token": {
14
+ "content": "</s>",
15
+ "lstrip": false,
16
+ "normalized": false,
17
+ "rstrip": false,
18
+ "single_word": false
19
+ },
20
+ "pad_token": {
21
+ "content": "<unk>",
22
+ "lstrip": false,
23
+ "normalized": false,
24
+ "rstrip": false,
25
+ "single_word": false
26
+ },
27
+ "unk_token": {
28
+ "content": "<unk>",
29
+ "lstrip": false,
30
+ "normalized": false,
31
+ "rstrip": false,
32
+ "single_word": false
33
+ }
34
+ }
tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
tokenizer.model ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9e556afd44213b6bd1be2b850ebbbd98f5481437a8021afaf58ee7fb1818d347
3
+ size 499723
tokenizer_config.json ADDED
@@ -0,0 +1,59 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "added_tokens_decoder": {
3
+ "0": {
4
+ "content": "<unk>",
5
+ "lstrip": false,
6
+ "normalized": false,
7
+ "rstrip": false,
8
+ "single_word": false,
9
+ "special": true
10
+ },
11
+ "1": {
12
+ "content": "<s>",
13
+ "lstrip": false,
14
+ "normalized": false,
15
+ "rstrip": false,
16
+ "single_word": false,
17
+ "special": true
18
+ },
19
+ "2": {
20
+ "content": "</s>",
21
+ "lstrip": false,
22
+ "normalized": false,
23
+ "rstrip": false,
24
+ "single_word": false,
25
+ "special": true
26
+ },
27
+ "32000": {
28
+ "content": "<fake_token_around_image>",
29
+ "lstrip": false,
30
+ "normalized": false,
31
+ "rstrip": false,
32
+ "single_word": false,
33
+ "special": true
34
+ },
35
+ "32001": {
36
+ "content": "<image>",
37
+ "lstrip": false,
38
+ "normalized": false,
39
+ "rstrip": false,
40
+ "single_word": false,
41
+ "special": true
42
+ }
43
+ },
44
+ "additional_special_tokens": [
45
+ "<image>",
46
+ "<fake_token_around_image>"
47
+ ],
48
+ "bos_token": "<s>",
49
+ "clean_up_tokenization_spaces": false,
50
+ "eos_token": "</s>",
51
+ "legacy": true,
52
+ "model_max_length": 2048,
53
+ "pad_token": "<unk>",
54
+ "sp_model_kwargs": {},
55
+ "spaces_between_special_tokens": false,
56
+ "tokenizer_class": "LlamaTokenizer",
57
+ "unk_token": "<unk>",
58
+ "use_default_system_prompt": true
59
+ }