Training in progress, step 40

Browse files

Files changed (9) hide show

axolotl_config.yaml +50 -0
config.json +31 -0
merges.txt +0 -0
model.safetensors +3 -0
special_tokens_map.json +30 -0
tokenizer.json +0 -0
tokenizer_config.json +33 -0
training_args.bin +3 -0
vocab.json +0 -0

axolotl_config.yaml ADDED Viewed

	@@ -0,0 +1,50 @@

+base_model: facebook/opt-125m
+batch_size: 128
+bf16: true
+chat_template: tokenizer_default_fallback_alpaca
+datasets:
+- format: custom
+  path: https://gradients.s3.eu-north-1.amazonaws.com/7575a7e5b38479e3_train_data.json?X-Amz-Algorithm=AWS4-HMAC-SHA256&X-Amz-Credential=AKIAVVZOOA7SA4UOFLPI%2F20250209%2Feu-north-1%2Fs3%2Faws4_request&X-Amz-Date=20250209T024133Z&X-Amz-Expires=604800&X-Amz-SignedHeaders=host&X-Amz-Signature=5b334fbc5cfa88197315bf2cd74f2114e46bdb6bf8716ba3c96b73b8c3e6ef9a
+  type:
+    field_instruction: instruction
+    field_output: output
+    format: '{instruction}'
+    no_input_format: '{instruction}'
+    system_format: '{system}'
+    system_prompt: ''
+device_map: auto
+eval_sample_packing: false
+eval_steps: 40
+flash_attention: true
+gradient_checkpointing: true
+group_by_length: true
+hub_model_id: SystemAdmin123/d778c1e6-0d20-4dd2-81e1-efc24a74b590
+hub_strategy: checkpoint
+learning_rate: 0.0002
+logging_steps: 10
+lr_scheduler: cosine
+max_steps: 10000
+micro_batch_size: 32
+model_type: AutoModelForCausalLM
+num_epochs: 100
+optimizer: adamw_bnb_8bit
+output_dir: /root/.sn56/axolotl/tmp/d778c1e6-0d20-4dd2-81e1-efc24a74b590
+pad_to_sequence_len: true
+resize_token_embeddings_to_32x: false
+sample_packing: true
+save_steps: 40
+save_total_limit: 2
+sequence_len: 2048
+tokenizer_type: GPT2TokenizerFast
+torch_dtype: bf16
+training_args_kwargs:
+  hub_private_repo: true
+trust_remote_code: true
+val_set_size: 0.1
+wandb_entity: ''
+wandb_mode: online
+wandb_name: facebook/opt-125m-https://gradients.s3.eu-north-1.amazonaws.com/7575a7e5b38479e3_train_data.json?X-Amz-Algorithm=AWS4-HMAC-SHA256&X-Amz-Credential=AKIAVVZOOA7SA4UOFLPI%2F20250209%2Feu-north-1%2Fs3%2Faws4_request&X-Amz-Date=20250209T024133Z&X-Amz-Expires=604800&X-Amz-SignedHeaders=host&X-Amz-Signature=5b334fbc5cfa88197315bf2cd74f2114e46bdb6bf8716ba3c96b73b8c3e6ef9a
+wandb_project: Gradients-On-Demand
+wandb_run: your_name
+wandb_runid: default
+warmup_ratio: 0.05

config.json ADDED Viewed

	@@ -0,0 +1,31 @@

+{
+  "_name_or_path": "facebook/opt-125m",
+  "_remove_final_layer_norm": false,
+  "activation_dropout": 0.0,
+  "activation_function": "relu",
+  "architectures": [
+    "OPTForCausalLM"
+  ],
+  "attention_dropout": 0.0,
+  "bos_token_id": 2,
+  "do_layer_norm_before": true,
+  "dropout": 0.1,
+  "enable_bias": true,
+  "eos_token_id": 2,
+  "ffn_dim": 3072,
+  "hidden_size": 768,
+  "init_std": 0.02,
+  "layer_norm_elementwise_affine": true,
+  "layerdrop": 0.0,
+  "max_position_embeddings": 2048,
+  "model_type": "opt",
+  "num_attention_heads": 12,
+  "num_hidden_layers": 12,
+  "pad_token_id": 1,
+  "prefix": "</s>",
+  "torch_dtype": "bfloat16",
+  "transformers_version": "4.47.1",
+  "use_cache": false,
+  "vocab_size": 50272,
+  "word_embed_proj_dim": 768
+}

merges.txt ADDED Viewed

The diff for this file is too large to render. See raw diff

model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:65b22bdf43d38b61cc704a197114972c4a1ec34ec00733695846e8aa16fe9346
+size 250501160

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,30 @@

+{
+  "bos_token": {
+    "content": "</s>",
+    "lstrip": false,
+    "normalized": true,
+    "rstrip": false,
+    "single_word": false
+  },
+  "eos_token": {
+    "content": "</s>",
+    "lstrip": false,
+    "normalized": true,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": {
+    "content": "<pad>",
+    "lstrip": false,
+    "normalized": true,
+    "rstrip": false,
+    "single_word": false
+  },
+  "unk_token": {
+    "content": "</s>",
+    "lstrip": false,
+    "normalized": true,
+    "rstrip": false,
+    "single_word": false
+  }
+}

tokenizer.json ADDED Viewed

The diff for this file is too large to render. See raw diff

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,33 @@

+{
+  "add_bos_token": true,
+  "add_prefix_space": false,
+  "added_tokens_decoder": {
+    "1": {
+      "content": "<pad>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "2": {
+      "content": "</s>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "bos_token": "</s>",
+  "chat_template": "{% for message in messages %}{% if message['role'] == 'user' %}{{ '### Instruction: ' + message['content'] + '\n\n' }}{% elif message['role'] == 'assistant' %}{{ '### Response: ' + message['content'] + eos_token}}{% endif %}{% endfor %}",
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "</s>",
+  "errors": "replace",
+  "extra_special_tokens": {},
+  "model_max_length": 1000000000000000019884624838656,
+  "pad_token": "<pad>",
+  "tokenizer_class": "GPT2Tokenizer",
+  "unk_token": "</s>",
+  "use_fast": true
+}

training_args.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:edcbfeaed8172c5acebdf55d3b1d6e875f976c0ce9e9f35f87ebdee1bf4f0252
+size 7032

vocab.json ADDED Viewed

The diff for this file is too large to render. See raw diff