Upload 7 files

Browse files

Files changed (7) hide show

README.md +50 -3
config.json +1 -0
pytorch_model.bin +3 -0
special_tokens_map.json +30 -0
test_metrics.json +1 -0
tokenizer_config.json +44 -0
vocab.txt +27 -0

README.md CHANGED Viewed

@@ -1,3 +1,50 @@
----
-license: cc-by-nc-sa-4.0
----

+---
+license: cc-by-nc-sa-4.0
+widget:
+- text: GCGACTCCGCCGCCCCGATCTCCCCGTCGTCCTACAGTGCTCTCCACATCGTAGGCGACCTGGTTGGACTCCTCGACGCCTTGTCCCTACCGCAGGTGTTTGTGGTGGGACAAGGCTGGGGAGCCCTGCTGGCGTGGAACCTCTGCATGTTCCGCCCCGAGCGGGTGCGCGCGCTGGTCAACATGAGCGTCGCCTTCATGCCGCGCAACCCCTCCGTGAAGCCACTTGAGTTGTTTCGGCGGCTCTACGGCGACGGATACTACCTCCTCCGGCTGCAGGAAC
+tags:
+- DNA
+- biology
+- genomics
+---
+# Plant foundation DNA large language models
+The plant DNA large language models (LLMs) contain a series of foundation models based on different model architectures, which are pre-trained on various plant reference genomes.
+All the models have a comparable model size between 90 MB and 150 MB, BPE tokenizer is used for tokenization and 8000 tokens are included in the vocabulary.
+**Developed by:** zhangtaolab
+### Model Sources
+- **Repository:** [Plant DNA LLMs](https://github.com/zhangtaolab/plant_DNA_LLMs)
+- **Manuscript:** [Versatile applications of foundation DNA language models in plant genomes]()
+### Architecture
+The model is trained based on the State-Space Mamba-130m model with modified tokenizer specific for DNA sequence.
+This model is fine-tuned for predicting H3K4me3 histone modification.
+### How to use
+Install the runtime library first:
+```bash
+pip install transformers
+pip install causal-conv1d<=1.2.0
+pip install mamba-ssm<2.0.0
+```
+Since `transformers` library (version < 4.43.0) does not provide a MambaForSequenceClassification function, we wrote a script to train Mamba model for sequence classification.
+An inference code can be found in our [GitHub](https://github.com/zhangtaolab/plant_DNA_LLMs).
+Note that Plant DNAMamba model requires NVIDIA GPU to run.
+### Training data
+We use a custom MambaForSequenceClassification script to fine-tune the model.
+Detailed training procedure can be found in our manuscript.
+#### Hardware
+Model was trained on a NVIDIA GTX4090 GPU (24 GB).

config.json ADDED Viewed

	@@ -0,0 +1 @@


1	+ {"d_model": 768, "n_layer": 24, "vocab_size": 27, "ssm_cfg": {}, "rms_norm": true, "residual_in_fp32": true, "fused_add_norm": true, "pad_vocab_size_multiple": 1, "tie_embeddings": true}

pytorch_model.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:25965e755c8809a8a933f15dd600d630188d28f0fe523715e3a5cf2dc436b823
+size 362263066

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,30 @@

+{
+  "cls_token": {
+    "content": "<cls>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "mask_token": {
+    "content": "<mask>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": {
+    "content": "<pad>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "unk_token": {
+    "content": "<unk>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  }
+}

test_metrics.json ADDED Viewed

	@@ -0,0 +1 @@


1	+ {'test_loss': 0.5903248190879822, 'test_accuracy': 0.86259765625, 'test_f1': 0.8563552833078101, 'test_precision': 0.8971122994652406, 'test_recall': 0.819140625, 'test_matthews_correlation': 0.7279500115916328, 'test_runtime': 32.1715, 'test_samples_per_second': 318.294, 'test_steps_per_second': 19.893}

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,44 @@

+{
+  "added_tokens_decoder": {
+    "0": {
+      "content": "<unk>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "1": {
+      "content": "<pad>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "2": {
+      "content": "<mask>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "3": {
+      "content": "<cls>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "clean_up_tokenization_spaces": true,
+  "cls_token": "<cls>",
+  "eos_token": null,
+  "mask_token": "<mask>",
+  "model_max_length": 512,
+  "pad_token": "<pad>",
+  "tokenizer_class": "EsmTokenizer",
+  "unk_token": "<unk>"
+}

vocab.txt ADDED Viewed

	@@ -0,0 +1,27 @@

+<unk>
+<pad>
+<mask>
+<cls>
+AA
+AT
+AC
+AG
+TA
+TT
+TC
+TG
+CA
+CT
+CC
+CG
+GA
+GT
+GC
+GG
+A
+T
+C
+G
+N
+<eos>
+<bos>