Training in progress, step 200

Browse files

Files changed (10) hide show

README.md +75 -0
config.json +33 -0
generation_config.json +6 -0
model.safetensors +3 -0
modeling_bit_llama.py +79 -0
special_tokens_map.json +24 -0
tokenizer.json +0 -0
tokenizer.model +3 -0
tokenizer_config.json +43 -0
training_args.bin +3 -0

README.md ADDED Viewed

	@@ -0,0 +1,75 @@

+---
+tags:
+- generated_from_trainer
+model-index:
+- name: myBit-Llama2-jp-127M-3
+  results: []
+---
+<!-- This model card has been generated automatically according to the information the Trainer had access to. You
+should probably proofread and complete it, then remove this comment. -->
+# myBit-Llama2-jp-127M-3
+This model is a fine-tuned version of [](https://huggingface.co/) on an unknown dataset.
+It achieves the following results on the evaluation set:
+- Loss: 3.8396
+## Model description
+More information needed
+## Intended uses & limitations
+More information needed
+## Training and evaluation data
+More information needed
+## Training procedure
+### Training hyperparameters
+The following hyperparameters were used during training:
+- learning_rate: 0.00024
+- train_batch_size: 96
+- eval_batch_size: 96
+- seed: 42
+- optimizer: Adam with betas=(0.9,0.999) and epsilon=1e-08
+- lr_scheduler_type: polynomial
+- lr_scheduler_warmup_steps: 5000
+- num_epochs: 1
+### Training results
+| Training Loss | Epoch | Step  | Validation Loss |
+|:-------------:|:-----:|:-----:|:---------------:|
+| 6.9048        | 0.05  | 2000  | 4.7602          |
+| 4.4421        | 0.1   | 4000  | 4.2117          |
+| 4.0625        | 0.15  | 6000  | 3.9227          |
+| 3.807         | 0.2   | 8000  | 3.7181          |
+| 3.6547        | 0.25  | 10000 | 3.5929          |
+| 3.5296        | 0.29  | 12000 | 3.4812          |
+| 3.4492        | 0.34  | 14000 | 3.4236          |
+| 3.4065        | 0.39  | 16000 | 3.3923          |
+| 3.3816        | 0.44  | 18000 | 3.3778          |
+| 3.3815        | 0.49  | 20000 | 3.3907          |
+| 3.431         | 0.54  | 22000 | 3.4870          |
+| 3.5507        | 0.59  | 24000 | 3.5969          |
+| 3.6557        | 0.64  | 26000 | 3.6918          |
+| 3.715         | 0.69  | 28000 | 3.7377          |
+| 3.7646        | 0.74  | 30000 | 3.7620          |
+| 3.8005        | 0.79  | 32000 | 3.8221          |
+| 3.8288        | 0.83  | 34000 | 3.8550          |
+| 3.8552        | 0.88  | 36000 | 3.8449          |
+| 3.8591        | 0.93  | 38000 | 3.8483          |
+| 3.8452        | 0.98  | 40000 | 3.8396          |
+### Framework versions
+- Transformers 4.38.2
+- Pytorch 2.2.1+cu121
+- Datasets 2.18.0
+- Tokenizers 0.15.2

config.json ADDED Viewed

	@@ -0,0 +1,33 @@

+{
+  "architectures": [
+    "BitLlamaForCausalLM"
+  ],
+  "attention_bias": false,
+  "attention_dropout": 0.0,
+  "auto_map": {
+    "AutoConfig": "modeling_bit_llama.BitLlamaConfig",
+    "AutoModelForCausalLM": "modeling_bit_llama.BitLlamaForCausalLM"
+  },
+  "bits": 8,
+  "bos_token_id": 1,
+  "eos_token_id": 2,
+  "hidden_act": "silu",
+  "hidden_size": 768,
+  "initializer_range": 0.02,
+  "intermediate_size": 1536,
+  "max_position_embeddings": 1024,
+  "model_type": "bit_llama",
+  "n_ctx": 128,
+  "num_attention_heads": 12,
+  "num_hidden_layers": 12,
+  "num_key_value_heads": 4,
+  "pretraining_tp": 1,
+  "rms_norm_eps": 1e-05,
+  "rope_scaling": null,
+  "rope_theta": 10000.0,
+  "tie_word_embeddings": false,
+  "torch_dtype": "float32",
+  "transformers_version": "4.39.1",
+  "use_cache": true,
+  "vocab_size": 43176
+}

generation_config.json ADDED Viewed

	@@ -0,0 +1,6 @@

+{
+  "_from_model_config": true,
+  "bos_token_id": 1,
+  "eos_token_id": 2,
+  "transformers_version": "4.38.2"
+}

model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:8751cbbbd70b51c14908672987f94e710682d7fdb3fbb4e6a083ce7e0f6b0989
+size 510960712

modeling_bit_llama.py ADDED Viewed

	@@ -0,0 +1,79 @@

+from typing import Optional
+from transformers.models.llama.modeling_llama import (
+    LlamaConfig,
+    LlamaModel,
+    LlamaForCausalLM,
+    LlamaAttention,
+    LlamaFlashAttention2,
+    LlamaSdpaAttention,
+    LlamaMLP,
+    LlamaDecoderLayer,
+)
+from mybitnet.bitnet import BitLinear
+from torch import nn
+class BitLlamaConfig(LlamaConfig):
+    model_type = "bit_llama"
+    def __init__(self, bits=8, **kwargs):
+        super().__init__(**kwargs)
+        self.bits = bits
+class BitLlamaMLP(LlamaMLP):
+    def __init__(self, config):
+        super().__init__(config)
+        self.gate_proj = BitLinear(self.hidden_size, self.intermediate_size, bias=False, bits=config.bits, flg_before_linear=True)
+        self.up_proj = BitLinear(self.hidden_size, self.intermediate_size, bias=False, bits=config.bits, flg_before_linear=True)
+        self.down_proj = BitLinear(self.intermediate_size, self.hidden_size, bias=False, bits=config.bits, flg_before_linear=False)
+class BitLlamaAttention(LlamaAttention):
+    def __init__(self, config: BitLlamaConfig, layer_idx: Optional[int] = None):
+        super().__init__(config)
+        self.q_proj = BitLinear(self.hidden_size, self.num_heads * self.head_dim, bias=False, bits=config.bits, flg_before_linear=True)
+        self.k_proj = BitLinear(self.hidden_size, self.num_key_value_heads * self.head_dim, bias=False, bits=config.bits, flg_before_linear=True)
+        self.v_proj = BitLinear(self.hidden_size, self.num_key_value_heads * self.head_dim, bias=False, bits=config.bits, flg_before_linear=True)
+        self.o_proj = BitLinear(self.hidden_size, self.hidden_size, bias=False, bits=config.bits, flg_before_linear=True)
+class BitLlamaFlashAttention2(LlamaFlashAttention2):
+    def __init__(self, config: BitLlamaConfig, layer_idx: Optional[int] = None):
+        super().__init__(config, layer_idx)
+        self.q_proj = BitLinear(self.hidden_size, self.num_heads * self.head_dim, bias=False, bits=config.bits, flg_before_linear=True)
+        self.k_proj = BitLinear(self.hidden_size, self.num_key_value_heads * self.head_dim, bias=False, bits=config.bits, flg_before_linear=True)
+        self.v_proj = BitLinear(self.hidden_size, self.num_key_value_heads * self.head_dim, bias=False, bits=config.bits, flg_before_linear=True)
+        self.o_proj = BitLinear(self.hidden_size, self.hidden_size, bias=False, bits=config.bits, flg_before_linear=True)
+class BitLlamaSdpaAttention(LlamaSdpaAttention):
+    def __init__(self, config: BitLlamaConfig, layer_idx: Optional[int] = None):
+        super().__init__(config, layer_idx)
+        self.q_proj = BitLinear(self.hidden_size, self.num_heads * self.head_dim, bias=False, bits=config.bits, flg_before_linear=True)
+        self.k_proj = BitLinear(self.hidden_size, self.num_key_value_heads * self.head_dim, bias=False, bits=config.bits, flg_before_linear=True)
+        self.v_proj = BitLinear(self.hidden_size, self.num_key_value_heads * self.head_dim, bias=False, bits=config.bits, flg_before_linear=True)
+        self.o_proj = BitLinear(self.hidden_size, self.hidden_size, bias=False, bits=config.bits, flg_before_linear=True)
+BITLLAMA_ATTENTION_CLASSES = {
+    "eager": BitLlamaAttention,
+    "flash_attention_2": BitLlamaFlashAttention2,
+    "sdpa": BitLlamaSdpaAttention,
+}
+class BitLlamaDecoderLayer(LlamaDecoderLayer):
+    def __init__(self, config: BitLlamaConfig, layer_idx: int):
+        super().__init__(config, layer_idx)
+        self.self_attn = BITLLAMA_ATTENTION_CLASSES[config._attn_implementation](config=config, layer_idx=layer_idx)
+        self.mlp = BitLlamaMLP(config)
+class BitLlamaModel(LlamaModel):
+    def __init__(self, config: BitLlamaConfig):
+        super().__init__(config)
+        self.layers = nn.ModuleList(
+            [BitLlamaDecoderLayer(config, layer_idx) for layer_idx in range(config.num_hidden_layers)]
+        )
+class BitLlamaForCausalLM(LlamaForCausalLM):
+    config_class = BitLlamaConfig
+    def __init__(self, config: BitLlamaConfig):
+        super().__init__(config)
+        self.model = BitLlamaModel(config)
+        self.lm_head = BitLinear(config.hidden_size, config.vocab_size, bias=False, bits=config.bits, flg_before_linear=True)

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,24 @@

+{
+  "bos_token": {
+    "content": "<s>",
+    "lstrip": false,
+    "normalized": true,
+    "rstrip": false,
+    "single_word": false
+  },
+  "eos_token": {
+    "content": "</s>",
+    "lstrip": false,
+    "normalized": true,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": "</s>",
+  "unk_token": {
+    "content": "<unk>",
+    "lstrip": false,
+    "normalized": true,
+    "rstrip": false,
+    "single_word": false
+  }
+}

tokenizer.json ADDED Viewed

The diff for this file is too large to render. See raw diff

tokenizer.model ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:c877c5ca885bad5c19d1b1706a2703f8b30de90f03c1f834f8bdb9faf79821e8
+size 914000

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,43 @@

+{
+  "add_bos_token": true,
+  "add_eos_token": false,
+  "add_prefix_space": true,
+  "added_tokens_decoder": {
+    "0": {
+      "content": "<unk>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "1": {
+      "content": "<s>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "2": {
+      "content": "</s>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "bos_token": "<s>",
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "</s>",
+  "legacy": false,
+  "model_max_length": 1000000000000000019884624838656,
+  "pad_token": "</s>",
+  "padding_side": "right",
+  "sp_model_kwargs": {},
+  "spaces_between_special_tokens": false,
+  "tokenizer_class": "LlamaTokenizer",
+  "unk_token": "<unk>",
+  "use_default_system_prompt": false
+}

training_args.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:2edef78613ec11331f1e86b427554b65d0fab164a2cbea1240516eba1e252b2b
+size 4920