Upload 7 files

Browse files

Files changed (7) hide show

README.md +50 -3
config.json +1 -0
pytorch_model.bin +3 -0
special_tokens_map.json +30 -0
test_metrics.json +1 -0
tokenizer_config.json +44 -0
vocab.txt +75 -0

README.md CHANGED Viewed

@@ -1,3 +1,50 @@
----
-license: cc-by-nc-sa-4.0
----

+---
+license: cc-by-nc-sa-4.0
+widget:
+- text: AGTCCAGTGGACGACCAGCCACGGCTCCGGTCTGTAGAACCATCGCGGAAACGGCTCGCAAAACTCTAAACAGCGCAAACGATGCGCGCGCCGAAGCAACCCGGCTCTACTTATAAAAACGTCCAACGGTGAGCACCGAGCAGCTACTACTCGTACTCCCCCCACCGATC
+tags:
+- DNA
+- biology
+- genomics
+---
+# Plant foundation DNA large language models
+The plant DNA large language models (LLMs) contain a series of foundation models based on different model architectures, which are pre-trained on various plant reference genomes.
+All the models have a comparable model size between 90 MB and 150 MB, BPE tokenizer is used for tokenization and 8000 tokens are included in the vocabulary.
+**Developed by:** zhangtaolab
+### Model Sources
+- **Repository:** [Plant DNA LLMs](https://github.com/zhangtaolab/plant_DNA_LLMs)
+- **Manuscript:** [Versatile applications of foundation DNA large language models in plant genomes]()
+### Architecture
+The model is trained based on the State-Space Mamba-130m model with modified tokenizer specific for DNA sequence.
+This model is fine-tuned for predicting promoter strength in maize protoplasts system.
+### How to use
+Install the runtime library first:
+```bash
+pip install transformers
+pip install causal-conv1d<=1.2.0
+pip install mamba-ssm<2.0.0
+```
+Since `transformers` library (version < 4.43.0) does not provide a MambaForSequenceClassification function, we wrote a script to train Mamba model for sequence classification.
+An inference code can be found in our [GitHub](https://github.com/zhangtaolab/plant_DNA_LLMs).
+Note that Plant DNAMamba model requires NVIDIA GPU to run.
+### Training data
+We use a custom MambaForSequenceClassification script to fine-tune the model.
+Detailed training procedure can be found in our manuscript.
+#### Hardware
+Model was trained on a NVIDIA GTX4090 GPU (24 GB).

config.json ADDED Viewed

	@@ -0,0 +1 @@


1	+ {"d_model": 768, "n_layer": 24, "vocab_size": 75, "ssm_cfg": {}, "rms_norm": true, "residual_in_fp32": true, "fused_add_norm": true, "pad_vocab_size_multiple": 1, "tie_embeddings": true}

pytorch_model.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:b8258f80ae287fac56985bc3764c84c6fcb642a853061097c1b550e3014cd613
+size 362407450

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,30 @@

+{
+  "cls_token": {
+    "content": "<cls>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "mask_token": {
+    "content": "<mask>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": {
+    "content": "<pad>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "unk_token": {
+    "content": "<unk>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  }
+}

test_metrics.json ADDED Viewed

	@@ -0,0 +1 @@


1	+ {'test_loss': 0.4641941785812378, 'test_r2': 0.697, 'test_spearmanr': 0.8175756459903686, 'test_runtime': 23.8454, 'test_samples_per_second': 318.51, 'test_steps_per_second': 19.92}

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,44 @@

+{
+  "added_tokens_decoder": {
+    "0": {
+      "content": "<unk>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "1": {
+      "content": "<pad>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "2": {
+      "content": "<mask>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "3": {
+      "content": "<cls>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "clean_up_tokenization_spaces": true,
+  "cls_token": "<cls>",
+  "eos_token": null,
+  "mask_token": "<mask>",
+  "model_max_length": 512,
+  "pad_token": "<pad>",
+  "tokenizer_class": "EsmTokenizer",
+  "unk_token": "<unk>"
+}

vocab.txt ADDED Viewed

	@@ -0,0 +1,75 @@

+<unk>
+<pad>
+<mask>
+<cls>
+AAA
+AAT
+AAC
+AAG
+ATA
+ATT
+ATC
+ATG
+ACA
+ACT
+ACC
+ACG
+AGA
+AGT
+AGC
+AGG
+TAA
+TAT
+TAC
+TAG
+TTA
+TTT
+TTC
+TTG
+TCA
+TCT
+TCC
+TCG
+TGA
+TGT
+TGC
+TGG
+CAA
+CAT
+CAC
+CAG
+CTA
+CTT
+CTC
+CTG
+CCA
+CCT
+CCC
+CCG
+CGA
+CGT
+CGC
+CGG
+GAA
+GAT
+GAC
+GAG
+GTA
+GTT
+GTC
+GTG
+GCA
+GCT
+GCC
+GCG
+GGA
+GGT
+GGC
+GGG
+A
+T
+C
+G
+N
+<eos>
+<bos>