Upload 11 files

Browse files

Files changed (11) hide show

cached_lm_GPT2TokenizerFast_128_alignment.text +0 -0
cached_lm_GPT2TokenizerFast_128_alignment.text.lock +0 -0
config.json +38 -0
generation_config.json +6 -0
merges.txt +0 -0
pytorch_model.bin +3 -0
special_tokens_map.json +5 -0
tokenizer.json +0 -0
tokenizer_config.json +9 -0
train.py +75 -0
vocab.json +0 -0

cached_lm_GPT2TokenizerFast_128_alignment.text ADDED Viewed

Binary file (980 kB). View file

cached_lm_GPT2TokenizerFast_128_alignment.text.lock ADDED Viewed

File without changes

config.json ADDED Viewed

	@@ -0,0 +1,38 @@

+{
+  "_name_or_path": "/Users/migueldeguzman/Desktop/gpt2xl_algos/falcon-1b/v5/",
+  "alibi": true,
+  "apply_residual_connection_post_layernorm": false,
+  "architectures": [
+    "FalconForCausalLM"
+  ],
+  "attention_dropout": 0.0,
+  "auto_map": {
+    "AutoConfig": "configuration_falcon.FalconConfig",
+    "AutoModel": "modeling_falcon.FalconModel",
+    "AutoModelForCausalLM": "modeling_falcon.FalconForCausalLM",
+    "AutoModelForQuestionAnswering": "modeling_falcon.FalconForQuestionAnswering",
+    "AutoModelForSequenceClassification": "modeling_falcon.FalconForSequenceClassification",
+    "AutoModelForTokenClassification": "modeling_falcon.FalconForTokenClassification"
+  },
+  "bias": true,
+  "bos_token_id": 1,
+  "eos_token_id": 2,
+  "hidden_dropout": 0.0,
+  "hidden_size": 2048,
+  "initializer_range": 0.02,
+  "layer_norm_epsilon": 1e-05,
+  "max_position_embeddings": 2048,
+  "model_type": "falcon",
+  "multi_query": false,
+  "new_decoder_architecture": false,
+  "num_attention_heads": 32,
+  "num_hidden_layers": 24,
+  "num_kv_heads": 32,
+  "parallel_attn": false,
+  "rope_scaling": null,
+  "rope_theta": 10000.0,
+  "torch_dtype": "float32",
+  "transformers_version": "4.33.3",
+  "use_cache": true,
+  "vocab_size": 50304
+}

generation_config.json ADDED Viewed

	@@ -0,0 +1,6 @@

+{
+  "_from_model_config": true,
+  "bos_token_id": 1,
+  "eos_token_id": 2,
+  "transformers_version": "4.33.3"
+}

merges.txt ADDED Viewed

The diff for this file is too large to render. See raw diff

pytorch_model.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:c2e590915624eaad4583dba527b9c17981ed2a15f308708f949fd9bcd65ad07f
+size 5246593815

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,5 @@

+{
+  "bos_token": "<|endoftext|>",
+  "eos_token": "<|endoftext|>",
+  "unk_token": "<|endoftext|>"
+}

tokenizer.json ADDED Viewed

The diff for this file is too large to render. See raw diff

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "add_prefix_space": false,
+  "bos_token": "<|endoftext|>",
+  "clean_up_tokenization_spaces": true,
+  "eos_token": "<|endoftext|>",
+  "model_max_length": 1024,
+  "tokenizer_class": "GPT2Tokenizer",
+  "unk_token": "<|endoftext|>"
+}

train.py ADDED Viewed

	@@ -0,0 +1,75 @@

+import os
+import sys
+import torch
+from transformers import AutoModelForCausalLM, AutoTokenizer, TextDataset, DataCollatorForLanguageModeling, Trainer, TrainingArguments
+class GPTAssistant:
+    def __init__(self, model_name="/Users/migueldeguzman/Desktop/gpt2xl_algos/falcon-1b/v5/"):  # Replace with your specific model
+        try:
+            # Load the tokenizer and model using the specified model name
+            self.tokenizer = AutoTokenizer.from_pretrained(model_name, trust_remote_code=True)
+            self.model = AutoModelForCausalLM.from_pretrained(model_name)
+        except Exception as e:
+            print(f"Error initializing the model or tokenizer: {e}")
+            sys.exit(1)
+    def fine_tune(self, answer_file_path, model_output_dir, epochs=1.0):
+        # Load dataset for training
+        try:
+            train_dataset = TextDataset(
+                tokenizer=self.tokenizer,
+                file_path=answer_file_path,
+                block_size=128
+            )
+        except Exception as e:
+            print(f"Error loading training dataset: {e}")
+            sys.exit(1)  # Exit the script if dataset loading fails
+        # Prepare data collator for language modeling
+        data_collator = DataCollatorForLanguageModeling(
+            tokenizer=self.tokenizer,
+            mlm=False
+        )
+        total_steps = len(train_dataset) * epochs
+        warmup_steps = 0.1 * total_steps
+        # Set training arguments
+        training_args = TrainingArguments(
+            output_dir=model_output_dir,
+            overwrite_output_dir=True,
+            num_train_epochs=epochs,
+            per_device_train_batch_size=4,
+            save_steps=10_000,
+            save_total_limit=2,
+            weight_decay=0.005,
+            gradient_accumulation_steps=8,
+            learning_rate=3e-6,
+            lr_scheduler_type='cosine',
+            warmup_steps=warmup_steps
+        )
+        # Initialize Trainer
+        trainer = Trainer(
+            model=self.model,
+            args=training_args,
+            data_collator=data_collator,
+            train_dataset=train_dataset
+        )
+        # Train and save the model
+        trainer.train()
+        self.model.save_pretrained(model_output_dir)
+        self.tokenizer.save_pretrained(model_output_dir)
+def main():
+    # Specify the file path for training data and output directory
+    text_file_path = "/Users/migueldeguzman/Desktop/gpt2xl_algos/falcon-1b/v6/alignment.text"  # Replace with your training data file path
+    model_output_dir = "/Users/migueldeguzman/Desktop/gpt2xl_algos/falcon-1b/v6/"  # Replace with your desired output directory
+    # Initialize GPTAssistant and fine-tune the model
+    assistant = GPTAssistant()
+    assistant.fine_tune(text_file_path, model_output_dir)
+if __name__ == "__main__":
+    main()

vocab.json ADDED Viewed

The diff for this file is too large to render. See raw diff